Files
CoopAllTheThings/tools/mock_game/render_dx10.cpp
BlackMark 8a43d2f568 Mock game: render every backend UNCAPPED + assert a healthy present rate
The mock backends presented with vsync ("a game-like cadence") -- wrong for a
perf/stress fixture: it does trivial work on an RTX 4090, so it must run as fast
as it can. Vsync capped them to tens of fps (dx9 30, dx10 23, dx11 63, dx12 126),
which hid both capture-induced slowdowns and the hook-removal race. Uncapped now:

  dx9/dx10 INTERVAL_IMMEDIATE / Present(0,0) (BLT), dx11/dx12 ALLOW_TEARING +
  Present(0, ALLOW_TEARING) (flip), gl wglSwapIntervalEXT(0), vk IMMEDIATE/MAILBOX.

Measured no-hook: dx9 ~21000, dx10 ~2800, dx11 ~17000, dx12 ~12000, gl ~26000, vk
~24000 fps.

mock_game_test now adds a present-rate floor per backend (>= 300/s while
capturing): with the hook live every backend stays in the hundreds-thousands
(vk 13500, dx11 9000+, gl 1800, dx9/10 ~1000-1600, dx12 2500). This is the
dimension the frame-advance checks missed -- the Vulkan 144->3 FPS stall still
advanced frames -- so it catches a present-thread stall OR an accidental vsync.

The faster storm exposed the hook-removal UAF fixed in the previous commit.
README roadmap + lessons-learned updated (incl. correcting the old "reset makes
in-flight trampoline calls safe" claim). Full suite 21/21.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
2026-06-23 09:45:00 +02:00

130 lines
4.1 KiB
C++

// DX10 backend for the mock game. DX10 has no rect-clear (ClearView), so it composes the
// animated pattern (background + moving bar + top-left frame-counter block) into a CPU buffer
// each frame, uploads it into a scratch texture, and CopyResource's that into the swap-chain
// back buffer -- no shaders. It presents through a real D3D10 DXGI swap chain, so the capture
// hook sees a genuine D3D10 present.
#include "render_backend.hpp"
#include <vector>
#include <d3d10.h>
#include <dxgi.h>
#include <wrl/client.h>
using Microsoft::WRL::ComPtr;
namespace coop::mock
{
namespace
{
std::uint32_t pack(std::uint8_t r, std::uint8_t g, std::uint8_t b, std::uint8_t a = 255)
{
// R8G8B8A8_UNORM byte order: R in the low byte.
return static_cast<std::uint32_t>(r) | (static_cast<std::uint32_t>(g) << 8) |
(static_cast<std::uint32_t>(b) << 16) | (static_cast<std::uint32_t>(a) << 24);
}
class Dx10Backend : public RenderBackend
{
public:
bool init(HWND hwnd, std::uint32_t width, std::uint32_t height) override
{
width_ = width;
height_ = height;
px_.resize(static_cast<std::size_t>(width) * height);
DXGI_SWAP_CHAIN_DESC desc = {};
desc.BufferDesc.Width = width;
desc.BufferDesc.Height = height;
desc.BufferDesc.Format = DXGI_FORMAT_R8G8B8A8_UNORM; // UNORM (not sRGB) so the frame code is exact
desc.BufferDesc.RefreshRate.Numerator = 60;
desc.BufferDesc.RefreshRate.Denominator = 1;
desc.SampleDesc.Count = 1;
desc.BufferUsage = DXGI_USAGE_RENDER_TARGET_OUTPUT;
desc.BufferCount = 2;
desc.OutputWindow = hwnd;
desc.Windowed = TRUE;
desc.SwapEffect = DXGI_SWAP_EFFECT_DISCARD; // blt model: GetBuffer(0) is the back buffer
if (FAILED(D3D10CreateDeviceAndSwapChain(nullptr, D3D10_DRIVER_TYPE_HARDWARE, nullptr, 0,
D3D10_SDK_VERSION, &desc, swap_.GetAddressOf(),
device_.GetAddressOf())))
{
return false;
}
// Scratch texture matching the back buffer; we upload the CPU frame into it then
// CopyResource it into the back buffer (DX10 can't fill a sub-rect of an RTV directly).
D3D10_TEXTURE2D_DESC td = {};
td.Width = width;
td.Height = height;
td.MipLevels = 1;
td.ArraySize = 1;
td.Format = DXGI_FORMAT_R8G8B8A8_UNORM;
td.SampleDesc.Count = 1;
td.Usage = D3D10_USAGE_DEFAULT;
td.BindFlags = D3D10_BIND_SHADER_RESOURCE;
return SUCCEEDED(device_->CreateTexture2D(&td, nullptr, scratch_.GetAddressOf()));
}
void render_and_present(std::uint32_t frame) override
{
const std::uint32_t bg = pack(static_cast<std::uint8_t>((frame * 2) % 256),
static_cast<std::uint8_t>((frame * 3) % 256),
static_cast<std::uint8_t>((frame * 5) % 256));
const std::uint32_t whitepx = pack(255, 255, 255);
std::uint8_t fr = 0, fg = 0, fb = 0;
frame_to_rgb(frame, fr, fg, fb);
const std::uint32_t code = pack(fr, fg, fb);
const std::uint32_t span = width_ > 24 ? width_ - 24 : 1;
const std::uint32_t bx = (frame * 4) % span; // moving vertical bar
for (std::uint32_t y = 0; y < height_; ++y)
{
std::uint32_t* row = px_.data() + static_cast<std::size_t>(y) * width_;
for (std::uint32_t x = 0; x < width_; ++x)
{
std::uint32_t c = bg;
if (x >= bx && x < bx + 24)
{
c = whitepx;
}
if (x < kFrameBlock && y < kFrameBlock)
{
c = code; // top-left frame-counter block
}
row[x] = c;
}
}
device_->UpdateSubresource(scratch_.Get(), 0, nullptr, px_.data(), static_cast<UINT>(width_ * 4), 0);
ComPtr<ID3D10Texture2D> back;
if (SUCCEEDED(swap_->GetBuffer(0, IID_PPV_ARGS(back.GetAddressOf()))))
{
device_->CopyResource(back.Get(), scratch_.Get());
}
swap_->Present(0, 0); // uncapped (BLT model): the mock does nothing -> it must be fast
}
[[nodiscard]] const char* name() const override
{
return "dx10";
}
private:
std::uint32_t width_ = 0;
std::uint32_t height_ = 0;
std::vector<std::uint32_t> px_;
ComPtr<ID3D10Device> device_;
ComPtr<IDXGISwapChain> swap_;
ComPtr<ID3D10Texture2D> scratch_;
};
} // namespace
std::unique_ptr<RenderBackend> create_dx10_backend()
{
return std::make_unique<Dx10Backend>();
}
} // namespace coop::mock