Recover a guessed stream's channels + bit depth by correlation too (step b)
Extends the two-path correlation from rate-only to the full layout, removing the "channels/bit-depth assumed = device" limitation. correlate_format tries each candidate de-interleaving (float32 / int16; mono..7.1) of the hook capture, runs the rate correlation per layout, and keeps whichever aligns with the loopback; a wrong de-interleaving is noise and won't. The catch: the hook can't know a guessed stream's real frame size, so its verify tap pads each render buffer to the device block -- which over-reads stale staging bytes for a stream with fewer channels/bits, scrambling the audio. So the tap is now self-describing: it prefixes each buffer with its frame count ([count][count*device_block bytes]), and the host strips the padding per candidate layout (take the real count*real_block of each chunk) before de-interleaving. - audio_correlate.hpp: ChunkedCapture + chunk-aware correlate_format + candidate layouts; absolute-margin confidence gate (the true layout scores ~1.0, a truly ambiguous alternative within ~0.001 -- 2ch@R == 1ch@2R for identical channels -- is correctly left unconfident). - audio_hook.cpp: chunked verify tap (free-space-checked so framing can't tear). - audio_format_verifier: parse chunks; recover_layout path. AudioMirror now corrects the full format. - audio_correlation_test: layout recovery from padded chunks (stereo float, 16-bit PCM, 5.1, mono). audio_verify_test gains scenario (b): 2ch on a multichannel endpoint with distinct per-channel content (new env-gated ToneSource mode) -> recovers ch=2/32-bit float end-to-end. Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
This commit is contained in:
@@ -6,6 +6,7 @@
|
||||
// rates (plus a capture-latency skew and a little noise), then assert correlate_rate() recovers the
|
||||
// true rate -- including the hard 44100-vs-48000 case the cadence method can misread. Pure header
|
||||
// logic, no device.
|
||||
#include <algorithm>
|
||||
#include <cmath>
|
||||
#include <cstdint>
|
||||
#include <cstdio>
|
||||
@@ -55,6 +56,118 @@ std::vector<float> capture(unsigned rate, double seconds, double t0, double nois
|
||||
return out;
|
||||
}
|
||||
|
||||
// Per-channel continuous signal: each channel carries genuinely different content (its own
|
||||
// frequency set), like a real stereo/surround stream. This is what disambiguates the channel
|
||||
// count -- with identical channels, 2ch@R and 1ch@2R produce the same bytes and are truly
|
||||
// indistinguishable (the confidence gate correctly rejects that case).
|
||||
double multi(double t, unsigned channel)
|
||||
{
|
||||
const double k = 1.0 + 0.37 * static_cast<double>(channel); // distinct frequency scale per channel
|
||||
const double chirp = std::sin(2.0 * kPi * (300.0 * k * t + 140.0 * t * t));
|
||||
return 0.5 * std::sin(2.0 * kPi * 221.0 * k * t) + 0.28 * std::sin(2.0 * kPi * 437.0 * k * t + 0.6) +
|
||||
0.22 * chirp;
|
||||
}
|
||||
|
||||
double multi_mono(double t, unsigned channels)
|
||||
{
|
||||
double sum = 0.0;
|
||||
for (unsigned c = 0; c < channels; ++c)
|
||||
{
|
||||
sum += multi(t, c);
|
||||
}
|
||||
return sum / channels;
|
||||
}
|
||||
|
||||
// Encode the multichannel signal to raw interleaved PCM bytes at the given layout/rate.
|
||||
std::vector<std::uint8_t> encode(unsigned rate, unsigned channels, unsigned bits, unsigned tag, double seconds)
|
||||
{
|
||||
const unsigned bps = bits / 8;
|
||||
const std::size_t frames = static_cast<std::size_t>(rate * seconds);
|
||||
std::vector<std::uint8_t> out(frames * channels * bps);
|
||||
for (std::size_t i = 0; i < frames; ++i)
|
||||
{
|
||||
const double t = static_cast<double>(i) / rate;
|
||||
for (unsigned c = 0; c < channels; ++c)
|
||||
{
|
||||
const double s = multi(t, c);
|
||||
std::uint8_t* p = out.data() + (i * channels + c) * bps;
|
||||
if (tag == coop::kWaveFormatFloat)
|
||||
{
|
||||
const float f = static_cast<float>(s);
|
||||
std::memcpy(p, &f, 4);
|
||||
}
|
||||
else
|
||||
{
|
||||
const std::int16_t v = static_cast<std::int16_t>(s * 30000.0);
|
||||
std::memcpy(p, &v, 2);
|
||||
}
|
||||
}
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
// Build the hook's self-describing chunked capture from clean audio: split into ~480-frame chunks
|
||||
// and pad each frame from real_block up to `stride` with GARBAGE -- exactly what the hook's verify
|
||||
// tap produces (it over-reads a guessed stream whose real layout has fewer channels/bits than the
|
||||
// device block). correlate_format must strip the padding per candidate layout.
|
||||
coop::ChunkedCapture make_chunks(unsigned rate, unsigned ch, unsigned bits, unsigned tag, unsigned stride,
|
||||
double seconds)
|
||||
{
|
||||
coop::ChunkedCapture cap;
|
||||
cap.stride = stride;
|
||||
const std::vector<std::uint8_t> clean = encode(rate, ch, bits, tag, seconds);
|
||||
const unsigned real_block = ch * (bits / 8);
|
||||
const std::size_t frames = clean.size() / real_block;
|
||||
std::mt19937 rng(123);
|
||||
std::uniform_int_distribution<int> garbage(0, 255);
|
||||
std::size_t f = 0;
|
||||
while (f < frames)
|
||||
{
|
||||
const unsigned count = static_cast<unsigned>(std::min<std::size_t>(480, frames - f));
|
||||
cap.counts.push_back(count);
|
||||
// The hook reads count*stride contiguous bytes: the count real frames first
|
||||
// (count*real_block bytes), then count*(stride-real_block) bytes of stale over-read.
|
||||
const std::uint8_t* src = clean.data() + f * real_block;
|
||||
cap.bytes.insert(cap.bytes.end(), src, src + static_cast<std::size_t>(count) * real_block);
|
||||
for (std::size_t p = 0; p < static_cast<std::size_t>(count) * (stride - real_block); ++p)
|
||||
{
|
||||
cap.bytes.push_back(static_cast<std::uint8_t>(garbage(rng)));
|
||||
}
|
||||
f += count;
|
||||
}
|
||||
return cap;
|
||||
}
|
||||
|
||||
// One layout scenario: the hook bytes are at (true_*) and still being measured; the loopback is the
|
||||
// post-mix mono of the same audio at device_rate. Assert correlate_format recovers the full layout.
|
||||
void test_layout(unsigned true_rate, unsigned true_ch, unsigned true_bits, unsigned true_tag,
|
||||
unsigned device_rate, const char* label)
|
||||
{
|
||||
std::printf("== layout: %s (%u Hz / %u ch / %u-bit %s -> device %u Hz) ==\n", label, true_rate, true_ch,
|
||||
true_bits, true_tag == coop::kWaveFormatFloat ? "float" : "pcm", device_rate);
|
||||
// Device block 32 (8ch float) is the largest stride; every test layout's real block is <= 32.
|
||||
const coop::ChunkedCapture hook = make_chunks(true_rate, true_ch, true_bits, true_tag, /*stride=*/32, 0.6);
|
||||
// Loopback: the post-mix mono of the same audio, at the device rate, started ~18 ms later + noise.
|
||||
const std::size_t loop_frames = static_cast<std::size_t>(device_rate * 0.6);
|
||||
std::vector<float> loop(loop_frames);
|
||||
std::mt19937 rng(5);
|
||||
std::uniform_real_distribution<float> j(-1.0f, 1.0f);
|
||||
for (std::size_t m = 0; m < loop_frames; ++m)
|
||||
{
|
||||
loop[m] = static_cast<float>(multi_mono(0.018 + static_cast<double>(m) / device_rate, true_ch)) +
|
||||
0.02f * j(rng);
|
||||
}
|
||||
|
||||
const FormatCorrelation r =
|
||||
correlate_format(hook, loop, device_rate, standard_audio_rates(), standard_audio_layouts());
|
||||
std::printf(" picked %u Hz / %u ch / %u-bit %s score=%.3f runner_up=%.3f ok=%d\n", r.rate, r.channels,
|
||||
r.bits, r.tag == coop::kWaveFormatFloat ? "float" : "pcm", r.score, r.runner_up, r.ok ? 1 : 0);
|
||||
check(r.ok, "layout pick is confident");
|
||||
check(r.rate == true_rate, "recovered the true rate");
|
||||
check(r.channels == true_ch, "recovered the true channel count");
|
||||
check(r.bits == true_bits && r.tag == true_tag, "recovered the true bit depth / sample format");
|
||||
}
|
||||
|
||||
// One scenario: true hook rate `true_rate` mixed to `device_rate`. Assert the correlator picks
|
||||
// true_rate confidently and that the runner-up is clearly behind.
|
||||
void test_case(unsigned true_rate, unsigned device_rate, const char* label)
|
||||
@@ -84,6 +197,13 @@ int main()
|
||||
test_case(32000, 44100, "low-rate stream on a 44100 endpoint");
|
||||
test_case(48000, 44100, "48000 stream on a 44100 endpoint");
|
||||
|
||||
// Step (b): recover the full layout (channels + bit depth) when it differs from the device, by
|
||||
// trying candidate de-interleavings -- the case the rate-only step can't handle.
|
||||
test_layout(44100, 2, 32, coop::kWaveFormatFloat, 48000, "stereo float, wrong rate");
|
||||
test_layout(44100, 2, 16, coop::kWaveFormatPcm, 48000, "stereo 16-bit PCM (bit depth differs)");
|
||||
test_layout(48000, 6, 32, coop::kWaveFormatFloat, 48000, "5.1 float (channels differ)");
|
||||
test_layout(44100, 1, 32, coop::kWaveFormatFloat, 48000, "mono"); // 2ch@22050 isn't a candidate -> unambiguous
|
||||
|
||||
// Downmix sanity: a stereo interleaved buffer collapses to the same mono the scalar path uses.
|
||||
{
|
||||
std::printf("== downmix stereo -> mono ==\n");
|
||||
|
||||
Reference in New Issue
Block a user