audio: stereo downmix, volume boost options (#722)

## Link to GitHub Issue or related Pull Request, if one exists
Fixes #717, fixes #647

## Description of change
Adds an option to downmix surround sound (5.1, 7.1, etc) down to stereo
(2 speakers). This can be used in most WASAPI games, including Gitadora
and FTT.

How it works: when the game tries to open surround format (say, 7.1) we
create a fake buffer and tell the game that it is supported. In reality
we open a 2-channel stream with the real sound card. When the stream
begins, we downmix the channels down to two (using one of many
algorithms) and output to the sound card.

While we're here, implement an option to boost the game volume by some
decibel value, which is a feature often requested by SDVX players. This
was needed since downmixing can sometimes result in quieter audio.

## Testing
Tested GW Delta and FTT.

Downmix doesn't work on IIDX (32 bits) but volume boost works.
This commit is contained in:
bicarus
2026-05-31 20:49:22 -07:00
committed by GitHub
parent 6bb6b0a301
commit 84ea3f9e8e
13 changed files with 768 additions and 34 deletions
@@ -1,10 +1,72 @@
#include "audio_render_client.h"
#include <algorithm>
#include <cmath>
#include <cstdint>
#include <cstring>
#include "audio_client.h"
#include "hooks/audio/audio.h"
#include "wasapi_private.h"
const char CLASS_NAME[] = "WrappedIAudioRenderClient";
// scale every sample of an interleaved device buffer by `gain`, clamped to the format's range.
// supports 16/24/32-bit PCM and 32-bit float; other formats are left untouched.
static void apply_gain(BYTE *buffer, UINT32 frames, const WAVEFORMATEXTENSIBLE &fmt, float gain) {
const WAVEFORMATEX &f = fmt.Format;
const size_t samples = (size_t) frames * f.nChannels;
// KSDATAFORMAT_SUBTYPE_IEEE_FLOAT has Data1 == 3, _PCM has Data1 == 1
bool is_float = f.wFormatTag == WAVE_FORMAT_IEEE_FLOAT
|| (f.wFormatTag == WAVE_FORMAT_EXTENSIBLE && fmt.SubFormat.Data1 == 0x00000003);
if (is_float && f.wBitsPerSample == 32) {
auto p = reinterpret_cast<float *>(buffer);
for (size_t i = 0; i < samples; i++) {
p[i] = std::clamp(p[i] * gain, -1.0f, 1.0f);
}
return;
}
switch (f.wBitsPerSample) {
case 16: {
auto p = reinterpret_cast<int16_t *>(buffer);
for (size_t i = 0; i < samples; i++) {
p[i] = (int16_t) std::clamp((int) std::lround(p[i] * gain), -32768, 32767);
}
break;
}
case 24: {
// packed 24-bit little-endian
for (size_t i = 0; i < samples; i++) {
BYTE *s = buffer + i * 3;
int32_t v = s[0] | (s[1] << 8) | (s[2] << 16);
if (v & 0x800000) {
v |= ~0xFFFFFF; // sign extend
}
int64_t scaled = std::clamp<int64_t>(
std::llround((double) v * gain), -8388608, 8388607);
s[0] = scaled & 0xFF;
s[1] = (scaled >> 8) & 0xFF;
s[2] = (scaled >> 16) & 0xFF;
}
break;
}
case 32: {
auto p = reinterpret_cast<int32_t *>(buffer);
for (size_t i = 0; i < samples; i++) {
p[i] = (int32_t) std::clamp(
std::llround((double) p[i] * gain),
(long long) INT32_MIN, (long long) INT32_MAX);
}
break;
}
default:
break;
}
}
HRESULT STDMETHODCALLTYPE WrappedIAudioRenderClient::QueryInterface(REFIID riid, void **ppvObj) {
if (ppvObj == nullptr) {
return E_POINTER;
@@ -51,6 +113,12 @@ HRESULT STDMETHODCALLTYPE WrappedIAudioRenderClient::GetBuffer(UINT32 NumFramesR
return S_OK;
}
// surround downmix: reserve the real (stereo) device buffer, but hand the game a
// multi-channel scratch buffer that we downmix on release
if (this->client->downmix.enabled) {
CHECK_RESULT(this->client->downmix.get_buffer(pReal, NumFramesRequested, ppData));
}
// call original
HRESULT ret = pReal->GetBuffer(NumFramesRequested, ppData);
@@ -75,14 +143,44 @@ HRESULT STDMETHODCALLTYPE WrappedIAudioRenderClient::ReleaseBuffer(UINT32 NumFra
return S_OK;
}
// fix for audio pop effect
if (this->buffers_to_mute > 0 && this->client->frame_size > 0) {
// resolve the real device buffer for whichever path produced the audio
BYTE *device_buffer;
if (this->client->downmix.enabled) {
// zero out = mute
memset(this->audio_buffer, 0, NumFramesWritten * this->client->frame_size);
// downmix the game's multi-channel scratch into the real stereo buffer held since GetBuffer
this->client->downmix.write_device_buffer(NumFramesWritten, dwFlags);
device_buffer = this->client->downmix.current_buffer();
} else {
device_buffer = this->audio_buffer;
this->buffers_to_mute--;
// mute the first few buffers to avoid a startup pop
if (this->buffers_to_mute > 0 && this->client->frame_size > 0) {
memset(this->audio_buffer, 0, NumFramesWritten * this->client->frame_size);
this->buffers_to_mute--;
}
}
CHECK_RESULT(pReal->ReleaseBuffer(NumFramesWritten, dwFlags));
// boost the final output volume just before it reaches the device, layout-agnostic
if (hooks::audio::VOLUME_BOOST != 1.0f
&& device_buffer != nullptr
&& (dwFlags & AUDCLNT_BUFFERFLAGS_SILENT) == 0) {
static std::once_flag boost_printed;
std::call_once(boost_printed, []() {
log_info("audio::wasapi", "volume boost active: gain={}", hooks::audio::VOLUME_BOOST);
});
apply_gain(device_buffer, NumFramesWritten, this->client->device_format,
hooks::audio::VOLUME_BOOST);
}
HRESULT ret = pReal->ReleaseBuffer(NumFramesWritten, dwFlags);
if (this->client->downmix.enabled) {
this->client->downmix.buffer_released();
}
if (FAILED(ret)) {
PRINT_FAILED_RESULT(CLASS_NAME, __func__, ret);
}
return ret;
}