Files
CoopAllTheThings/tests/audio_correlation_test.cpp
BlackMark 7ada550930 Recover a guessed stream's channels + bit depth by correlation too (step b)
Extends the two-path correlation from rate-only to the full layout, removing the
"channels/bit-depth assumed = device" limitation. correlate_format tries each
candidate de-interleaving (float32 / int16; mono..7.1) of the hook capture, runs
the rate correlation per layout, and keeps whichever aligns with the loopback;
a wrong de-interleaving is noise and won't.

The catch: the hook can't know a guessed stream's real frame size, so its verify
tap pads each render buffer to the device block -- which over-reads stale staging
bytes for a stream with fewer channels/bits, scrambling the audio. So the tap is
now self-describing: it prefixes each buffer with its frame count
([count][count*device_block bytes]), and the host strips the padding per candidate
layout (take the real count*real_block of each chunk) before de-interleaving.

- audio_correlate.hpp: ChunkedCapture + chunk-aware correlate_format + candidate
  layouts; absolute-margin confidence gate (the true layout scores ~1.0, a truly
  ambiguous alternative within ~0.001 -- 2ch@R == 1ch@2R for identical channels --
  is correctly left unconfident).
- audio_hook.cpp: chunked verify tap (free-space-checked so framing can't tear).
- audio_format_verifier: parse chunks; recover_layout path. AudioMirror now corrects
  the full format.
- audio_correlation_test: layout recovery from padded chunks (stereo float, 16-bit
  PCM, 5.1, mono). audio_verify_test gains scenario (b): 2ch on a multichannel
  endpoint with distinct per-channel content (new env-gated ToneSource mode) ->
  recovers ch=2/32-bit float end-to-end.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
2026-06-23 02:50:59 +02:00

242 lines
10 KiB
C++

// Unit test for the two-path audio-format correlator (common/include/coop/audio_correlate.hpp).
//
// Models the real situation: the same game audio is captured twice -- by the render-hook at the
// stream's true (unknown) rate, and by process-loopback at the known device rate (the hook signal
// resampled by WASAPI's AUTOCONVERTPCM). We synthesize one continuous signal and sample it at both
// rates (plus a capture-latency skew and a little noise), then assert correlate_rate() recovers the
// true rate -- including the hard 44100-vs-48000 case the cadence method can misread. Pure header
// logic, no device.
#include <algorithm>
#include <cmath>
#include <cstdint>
#include <cstdio>
#include <random>
#include <vector>
#include "coop/audio_correlate.hpp"
using namespace coop;
namespace
{
int g_failures = 0;
void check(bool ok, const char* what)
{
std::printf("%s %s\n", ok ? " ok:" : "FAIL:", what);
if (!ok)
{
++g_failures;
}
}
constexpr double kPi = 3.14159265358979323846;
// A non-periodic, correlation-friendly continuous signal s(t): a couple of incommensurate tones
// plus a slow chirp, so cross-correlation has a single sharp peak (unlike a pure sine).
double source(double t)
{
const double chirp = std::sin(2.0 * kPi * (300.0 * t + 140.0 * t * t));
return 0.5 * std::sin(2.0 * kPi * 221.0 * t) + 0.28 * std::sin(2.0 * kPi * 437.0 * t + 0.6) +
0.22 * chirp;
}
// Sample s(t) at `rate` for `seconds`, starting at t0 (capture-latency skew), optionally adding
// white noise of amplitude `noise` (the post-mix path is not a bit-identical copy).
std::vector<float> capture(unsigned rate, double seconds, double t0, double noise, std::uint32_t seed)
{
const std::size_t n = static_cast<std::size_t>(rate * seconds);
std::vector<float> out(n);
std::mt19937 rng(seed);
std::uniform_real_distribution<float> jitter(-1.0f, 1.0f);
for (std::size_t i = 0; i < n; ++i)
{
const double t = t0 + static_cast<double>(i) / static_cast<double>(rate);
out[i] = static_cast<float>(source(t)) + static_cast<float>(noise) * jitter(rng);
}
return out;
}
// Per-channel continuous signal: each channel carries genuinely different content (its own
// frequency set), like a real stereo/surround stream. This is what disambiguates the channel
// count -- with identical channels, 2ch@R and 1ch@2R produce the same bytes and are truly
// indistinguishable (the confidence gate correctly rejects that case).
double multi(double t, unsigned channel)
{
const double k = 1.0 + 0.37 * static_cast<double>(channel); // distinct frequency scale per channel
const double chirp = std::sin(2.0 * kPi * (300.0 * k * t + 140.0 * t * t));
return 0.5 * std::sin(2.0 * kPi * 221.0 * k * t) + 0.28 * std::sin(2.0 * kPi * 437.0 * k * t + 0.6) +
0.22 * chirp;
}
double multi_mono(double t, unsigned channels)
{
double sum = 0.0;
for (unsigned c = 0; c < channels; ++c)
{
sum += multi(t, c);
}
return sum / channels;
}
// Encode the multichannel signal to raw interleaved PCM bytes at the given layout/rate.
std::vector<std::uint8_t> encode(unsigned rate, unsigned channels, unsigned bits, unsigned tag, double seconds)
{
const unsigned bps = bits / 8;
const std::size_t frames = static_cast<std::size_t>(rate * seconds);
std::vector<std::uint8_t> out(frames * channels * bps);
for (std::size_t i = 0; i < frames; ++i)
{
const double t = static_cast<double>(i) / rate;
for (unsigned c = 0; c < channels; ++c)
{
const double s = multi(t, c);
std::uint8_t* p = out.data() + (i * channels + c) * bps;
if (tag == coop::kWaveFormatFloat)
{
const float f = static_cast<float>(s);
std::memcpy(p, &f, 4);
}
else
{
const std::int16_t v = static_cast<std::int16_t>(s * 30000.0);
std::memcpy(p, &v, 2);
}
}
}
return out;
}
// Build the hook's self-describing chunked capture from clean audio: split into ~480-frame chunks
// and pad each frame from real_block up to `stride` with GARBAGE -- exactly what the hook's verify
// tap produces (it over-reads a guessed stream whose real layout has fewer channels/bits than the
// device block). correlate_format must strip the padding per candidate layout.
coop::ChunkedCapture make_chunks(unsigned rate, unsigned ch, unsigned bits, unsigned tag, unsigned stride,
double seconds)
{
coop::ChunkedCapture cap;
cap.stride = stride;
const std::vector<std::uint8_t> clean = encode(rate, ch, bits, tag, seconds);
const unsigned real_block = ch * (bits / 8);
const std::size_t frames = clean.size() / real_block;
std::mt19937 rng(123);
std::uniform_int_distribution<int> garbage(0, 255);
std::size_t f = 0;
while (f < frames)
{
const unsigned count = static_cast<unsigned>(std::min<std::size_t>(480, frames - f));
cap.counts.push_back(count);
// The hook reads count*stride contiguous bytes: the count real frames first
// (count*real_block bytes), then count*(stride-real_block) bytes of stale over-read.
const std::uint8_t* src = clean.data() + f * real_block;
cap.bytes.insert(cap.bytes.end(), src, src + static_cast<std::size_t>(count) * real_block);
for (std::size_t p = 0; p < static_cast<std::size_t>(count) * (stride - real_block); ++p)
{
cap.bytes.push_back(static_cast<std::uint8_t>(garbage(rng)));
}
f += count;
}
return cap;
}
// One layout scenario: the hook bytes are at (true_*) and still being measured; the loopback is the
// post-mix mono of the same audio at device_rate. Assert correlate_format recovers the full layout.
void test_layout(unsigned true_rate, unsigned true_ch, unsigned true_bits, unsigned true_tag,
unsigned device_rate, const char* label)
{
std::printf("== layout: %s (%u Hz / %u ch / %u-bit %s -> device %u Hz) ==\n", label, true_rate, true_ch,
true_bits, true_tag == coop::kWaveFormatFloat ? "float" : "pcm", device_rate);
// Device block 32 (8ch float) is the largest stride; every test layout's real block is <= 32.
const coop::ChunkedCapture hook = make_chunks(true_rate, true_ch, true_bits, true_tag, /*stride=*/32, 0.6);
// Loopback: the post-mix mono of the same audio, at the device rate, started ~18 ms later + noise.
const std::size_t loop_frames = static_cast<std::size_t>(device_rate * 0.6);
std::vector<float> loop(loop_frames);
std::mt19937 rng(5);
std::uniform_real_distribution<float> j(-1.0f, 1.0f);
for (std::size_t m = 0; m < loop_frames; ++m)
{
loop[m] = static_cast<float>(multi_mono(0.018 + static_cast<double>(m) / device_rate, true_ch)) +
0.02f * j(rng);
}
const FormatCorrelation r =
correlate_format(hook, loop, device_rate, standard_audio_rates(), standard_audio_layouts());
std::printf(" picked %u Hz / %u ch / %u-bit %s score=%.3f runner_up=%.3f ok=%d\n", r.rate, r.channels,
r.bits, r.tag == coop::kWaveFormatFloat ? "float" : "pcm", r.score, r.runner_up, r.ok ? 1 : 0);
check(r.ok, "layout pick is confident");
check(r.rate == true_rate, "recovered the true rate");
check(r.channels == true_ch, "recovered the true channel count");
check(r.bits == true_bits && r.tag == true_tag, "recovered the true bit depth / sample format");
}
// One scenario: true hook rate `true_rate` mixed to `device_rate`. Assert the correlator picks
// true_rate confidently and that the runner-up is clearly behind.
void test_case(unsigned true_rate, unsigned device_rate, const char* label)
{
std::printf("== %s (true %u Hz -> device %u Hz) ==\n", label, true_rate, device_rate);
// The hook captures at the true rate; the loopback captures the same signal at the device
// rate, started ~22 ms later (capture skew) with a little measurement noise.
const std::vector<float> hook = capture(true_rate, 0.55, 0.0, 0.0, 1);
const std::vector<float> loop = capture(device_rate, 0.55, 0.022, 0.02, 7);
const RateCorrelation r = correlate_rate(hook, loop, device_rate, standard_audio_rates());
std::printf(" picked %u Hz score=%.3f runner_up=%.3f ok=%d\n", r.rate, r.score, r.runner_up,
r.ok ? 1 : 0);
check(r.rate == true_rate, "correlator picked the true rate");
check(r.ok, "pick is confident (clears threshold + beats runner-up)");
check(r.score > r.runner_up, "winner scores above the runner-up");
}
} // namespace
int main()
{
// The headline case: Godot/Brotato render 44100 while the endpoint mixes 48000 -- the cadence
// method can misread this, the correlator must not.
test_case(44100, 48000, "godot/brotato case");
test_case(48000, 48000, "rate matches device");
test_case(96000, 48000, "high-rate stream");
test_case(32000, 44100, "low-rate stream on a 44100 endpoint");
test_case(48000, 44100, "48000 stream on a 44100 endpoint");
// Step (b): recover the full layout (channels + bit depth) when it differs from the device, by
// trying candidate de-interleavings -- the case the rate-only step can't handle.
test_layout(44100, 2, 32, coop::kWaveFormatFloat, 48000, "stereo float, wrong rate");
test_layout(44100, 2, 16, coop::kWaveFormatPcm, 48000, "stereo 16-bit PCM (bit depth differs)");
test_layout(48000, 6, 32, coop::kWaveFormatFloat, 48000, "5.1 float (channels differ)");
test_layout(44100, 1, 32, coop::kWaveFormatFloat, 48000, "mono"); // 2ch@22050 isn't a candidate -> unambiguous
// Downmix sanity: a stereo interleaved buffer collapses to the same mono the scalar path uses.
{
std::printf("== downmix stereo -> mono ==\n");
std::vector<float> stereo = {1.0f, 3.0f, 2.0f, 4.0f, -1.0f, 1.0f};
std::vector<float> mono;
correlate_detail::downmix(stereo.data(), 3, 2, mono);
check(mono.size() == 3 && std::fabs(mono[0] - 2.0f) < 1e-6 && std::fabs(mono[1] - 3.0f) < 1e-6 &&
std::fabs(mono[2] - 0.0f) < 1e-6,
"stereo frames average to mono");
}
// A pure guess with no shared signal must NOT be reported confident (loopback is unrelated noise).
{
std::printf("== unrelated signals are not confidently matched ==\n");
const std::vector<float> hook = capture(44100, 0.5, 0.0, 0.0, 1);
std::vector<float> noise(static_cast<std::size_t>(48000 * 0.5));
std::mt19937 rng(99);
std::uniform_real_distribution<float> d(-1.0f, 1.0f);
for (float& x : noise)
{
x = d(rng);
}
const RateCorrelation r = correlate_rate(hook, noise, 48000, standard_audio_rates());
std::printf(" picked %u Hz score=%.3f ok=%d\n", r.rate, r.score, r.ok ? 1 : 0);
check(!r.ok, "unrelated loopback is not a confident match");
}
if (g_failures == 0)
{
std::printf("PASS audio_correlation_test\n");
return 0;
}
std::printf("FAILED audio_correlation_test (%d)\n", g_failures);
return 1;
}