Fix Vulkan capture perf collapse: read back off the present thread

Against Sphere Spectacle (144 FPS, runs without Steam) the implicit-layer
capture dropped the game to ~3 FPS. Measured cause (per-stage trace in the
layer): the read-back ran on the game's PRESENT THREAD and spent ~370 ms per
1080p frame -- not the GPU copy (~2 ms) but the CPU swizzle, because the staging
buffer was a plain HOST_VISIBLE|HOST_COHERENT type (write-combined / uncached on
a discrete GPU), where a scattered CPU read runs at PCIe latency. 3 captures/s =
the 3 FPS the user saw.

Test-first: tests/vk_capture_perf_test reproduces the stall as a deterministic
unit test (372 ms/present, ratio 1.0 -> FAIL via `--sync`), then proves the fix
(0.02 ms/present, byte-correct BGRA->RGBA, ratio ~0 -> PASS).

Fix: extract the near-identical read-back from vk_hook.cpp and coop_vk_layer.cpp
into one shared coop::hook::VkCapture that:
  * has the present thread only record + submit the copy (sub-ms) and return;
  * runs a dedicated reaper thread for the fence wait + swizzle + D3D upload, off
    the critical path, with a ring of in-flight slots (game never waits);
  * allocates HOST_CACHED staging (fast CPU read), invalidating when non-coherent;
  * throttles capture to ~150 Hz (a guest stream is <= the host refresh; no point
    mirroring an uncapped 400+ FPS game and burning reaper CPU).

Real-game A/B: present rate now matches the no-capture baseline (605->470 vs
593->405 over the same ramp) with the mirror at ~130 fps -- no measurable impact.

Also adds present-thread overhead guards to the other GPU backends' hook tests
(present_overhead.hpp): DX11 0.05 ms, DX12 0.34 ms, OpenGL 0.09 ms overhead, all
asserted < one 60 Hz frame, so any future synchronous-stall regression fails.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
This commit is contained in:
2026-06-23 08:19:47 +02:00
parent 60957775b7
commit 304857dcf0
13 changed files with 1564 additions and 799 deletions

View File

@@ -7,6 +7,7 @@ add_library(coop_hook SHARED
src/opengl_hook.cpp
src/d3d9_hook.cpp
src/vk_hook.cpp
src/vk_capture.cpp
src/mkb_hook.cpp
src/debug_log.cpp
src/hook_registry.cpp)

553
hook/src/vk_capture.cpp Normal file
View File

@@ -0,0 +1,553 @@
#include "vk_capture.hpp"
#include <cstring>
#include "coop/protocol.hpp"
#include "coop/shared_memory.hpp"
namespace coop::hook
{
namespace
{
double now_ms()
{
LARGE_INTEGER f, c;
QueryPerformanceFrequency(&f);
QueryPerformanceCounter(&c);
return 1000.0 * static_cast<double>(c.QuadPart) / static_cast<double>(f.QuadPart);
}
} // namespace
VkCapture::~VkCapture()
{
shutdown();
}
// Prefer HOST_CACHED memory: the reaper reads every byte of this buffer back on the CPU, and an
// uncached / write-combined mapping (what a plain HOST_VISIBLE|HOST_COHERENT type usually is on a
// discrete GPU) makes that read run at PCIe latency -- the original ~370 ms/frame stall. Order of
// preference: cached+coherent (fast read, no invalidate) > cached (fast read, needs invalidate) >
// coherent-only (the slow fallback, only if nothing cached is host-visible).
bool VkCapture::find_readback_memory(std::uint32_t type_bits, std::uint32_t& out_index, bool& out_coherent)
{
VkPhysicalDeviceMemoryProperties mp{};
m_fns.GetPhysicalDeviceMemoryProperties(m_phys, &mp);
const VkMemoryPropertyFlags vis = VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT;
const VkMemoryPropertyFlags cached = VK_MEMORY_PROPERTY_HOST_CACHED_BIT;
const VkMemoryPropertyFlags coherent = VK_MEMORY_PROPERTY_HOST_COHERENT_BIT;
int best = -1;
bool best_coherent = true;
int best_rank = -1;
for (std::uint32_t i = 0; i < mp.memoryTypeCount; ++i)
{
if ((type_bits & (1u << i)) == 0)
{
continue;
}
const VkMemoryPropertyFlags f = mp.memoryTypes[i].propertyFlags;
if ((f & vis) == 0)
{
continue;
}
int rank;
if ((f & cached) && (f & coherent))
{
rank = 3;
}
else if (f & cached)
{
rank = 2;
}
else if (f & coherent)
{
rank = 1;
}
else
{
continue; // host-visible but neither cached nor coherent: unusable for a CPU read-back
}
if (rank > best_rank)
{
best_rank = rank;
best = static_cast<int>(i);
best_coherent = (f & coherent) != 0;
}
}
if (best < 0)
{
return false;
}
out_index = static_cast<std::uint32_t>(best);
out_coherent = best_coherent;
return true;
}
bool VkCapture::ensure_slot_pool()
{
if (m_pool != VK_NULL_HANDLE)
{
return true;
}
if (m_queue == VK_NULL_HANDLE)
{
m_fns.GetDeviceQueue(m_device, m_qfam, 0, &m_queue);
}
VkCommandPoolCreateInfo pci{VK_STRUCTURE_TYPE_COMMAND_POOL_CREATE_INFO};
pci.flags = VK_COMMAND_POOL_CREATE_RESET_COMMAND_BUFFER_BIT;
pci.queueFamilyIndex = m_qfam;
if (m_fns.CreateCommandPool(m_device, &pci, nullptr, &m_pool) != VK_SUCCESS)
{
return false;
}
for (Slot& s : m_slots)
{
VkCommandBufferAllocateInfo ai{VK_STRUCTURE_TYPE_COMMAND_BUFFER_ALLOCATE_INFO};
ai.commandPool = m_pool;
ai.level = VK_COMMAND_BUFFER_LEVEL_PRIMARY;
ai.commandBufferCount = 1;
VkFenceCreateInfo fi{VK_STRUCTURE_TYPE_FENCE_CREATE_INFO};
VkSemaphoreCreateInfo si{VK_STRUCTURE_TYPE_SEMAPHORE_CREATE_INFO};
if (m_fns.AllocateCommandBuffers(m_device, &ai, &s.cmd) != VK_SUCCESS ||
m_fns.CreateFence(m_device, &fi, nullptr, &s.fence) != VK_SUCCESS ||
m_fns.CreateSemaphore(m_device, &si, nullptr, &s.present_sem) != VK_SUCCESS)
{
return false;
}
}
return true;
}
bool VkCapture::ensure_staging(Slot& s, std::uint32_t w, std::uint32_t h)
{
const VkDeviceSize need = static_cast<VkDeviceSize>(w) * h * 4;
if (s.staging != VK_NULL_HANDLE && s.size == need)
{
return true;
}
if (s.mapped != nullptr)
{
m_fns.UnmapMemory(m_device, s.mem);
s.mapped = nullptr;
}
if (s.staging != VK_NULL_HANDLE)
{
m_fns.DestroyBuffer(m_device, s.staging, nullptr);
s.staging = VK_NULL_HANDLE;
}
if (s.mem != VK_NULL_HANDLE)
{
m_fns.FreeMemory(m_device, s.mem, nullptr);
s.mem = VK_NULL_HANDLE;
}
s.size = 0;
VkBufferCreateInfo bci{VK_STRUCTURE_TYPE_BUFFER_CREATE_INFO};
bci.size = need;
bci.usage = VK_BUFFER_USAGE_TRANSFER_DST_BIT;
bci.sharingMode = VK_SHARING_MODE_EXCLUSIVE;
if (m_fns.CreateBuffer(m_device, &bci, nullptr, &s.staging) != VK_SUCCESS)
{
return false;
}
VkMemoryRequirements mr{};
m_fns.GetBufferMemoryRequirements(m_device, s.staging, &mr);
std::uint32_t mt = 0;
bool coherent = true;
if (!find_readback_memory(mr.memoryTypeBits, mt, coherent))
{
m_fns.DestroyBuffer(m_device, s.staging, nullptr);
s.staging = VK_NULL_HANDLE;
return false;
}
VkMemoryAllocateInfo mai{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO};
mai.allocationSize = mr.size;
mai.memoryTypeIndex = mt;
if (m_fns.AllocateMemory(m_device, &mai, nullptr, &s.mem) != VK_SUCCESS ||
m_fns.BindBufferMemory(m_device, s.staging, s.mem, 0) != VK_SUCCESS ||
m_fns.MapMemory(m_device, s.mem, 0, VK_WHOLE_SIZE, 0, &s.mapped) != VK_SUCCESS)
{
if (s.mem != VK_NULL_HANDLE)
{
m_fns.FreeMemory(m_device, s.mem, nullptr);
s.mem = VK_NULL_HANDLE;
}
m_fns.DestroyBuffer(m_device, s.staging, nullptr);
s.staging = VK_NULL_HANDLE;
return false;
}
s.size = need;
s.coherent = coherent;
return true;
}
void VkCapture::free_slots()
{
for (Slot& s : m_slots)
{
if (s.mapped != nullptr)
{
m_fns.UnmapMemory(m_device, s.mem);
s.mapped = nullptr;
}
if (s.staging != VK_NULL_HANDLE)
{
m_fns.DestroyBuffer(m_device, s.staging, nullptr);
s.staging = VK_NULL_HANDLE;
}
if (s.mem != VK_NULL_HANDLE)
{
m_fns.FreeMemory(m_device, s.mem, nullptr);
s.mem = VK_NULL_HANDLE;
}
if (s.present_sem != VK_NULL_HANDLE)
{
m_fns.DestroySemaphore(m_device, s.present_sem, nullptr);
s.present_sem = VK_NULL_HANDLE;
}
if (s.fence != VK_NULL_HANDLE)
{
m_fns.DestroyFence(m_device, s.fence, nullptr);
s.fence = VK_NULL_HANDLE;
}
s.size = 0;
s.busy.store(false, std::memory_order_relaxed);
}
if (m_pool != VK_NULL_HANDLE)
{
m_fns.DestroyCommandPool(m_device, m_pool, nullptr); // frees the command buffers
m_pool = VK_NULL_HANDLE;
}
}
// --- D3D11 shared texture (reaper thread only; shutdown releases after the reaper has joined) ------
bool VkCapture::ensure_d3d()
{
if (m_d3d != nullptr)
{
return true;
}
return SUCCEEDED(D3D11CreateDevice(nullptr, D3D_DRIVER_TYPE_HARDWARE, nullptr, 0, nullptr, 0,
D3D11_SDK_VERSION, &m_d3d, nullptr, &m_d3d_ctx)) &&
m_d3d != nullptr;
}
void VkCapture::release_shared()
{
if (m_shared_mutex != nullptr)
{
m_shared_mutex->Release();
m_shared_mutex = nullptr;
}
if (m_shared_tex != nullptr)
{
m_shared_tex->Release();
m_shared_tex = nullptr;
}
if (m_shared_handle != nullptr)
{
CloseHandle(m_shared_handle);
m_shared_handle = nullptr;
}
m_share_w = m_share_h = 0;
}
void VkCapture::release_d3d()
{
release_shared();
if (m_d3d_ctx != nullptr)
{
m_d3d_ctx->Release();
m_d3d_ctx = nullptr;
}
if (m_d3d != nullptr)
{
m_d3d->Release();
m_d3d = nullptr;
}
}
bool VkCapture::ensure_shared_texture(UINT w, UINT h)
{
if (m_shared_tex != nullptr && m_share_w == w && m_share_h == h)
{
return true;
}
release_shared();
D3D11_TEXTURE2D_DESC d{};
d.Width = w;
d.Height = h;
d.MipLevels = 1;
d.ArraySize = 1;
d.Format = DXGI_FORMAT_R8G8B8A8_UNORM;
d.SampleDesc.Count = 1;
d.Usage = D3D11_USAGE_DEFAULT;
d.BindFlags = D3D11_BIND_SHADER_RESOURCE;
d.MiscFlags = D3D11_RESOURCE_MISC_SHARED_NTHANDLE | D3D11_RESOURCE_MISC_SHARED_KEYEDMUTEX;
if (FAILED(m_d3d->CreateTexture2D(&d, nullptr, &m_shared_tex)) || m_shared_tex == nullptr)
{
return false;
}
IDXGIResource1* res = nullptr;
if (FAILED(m_shared_tex->QueryInterface(__uuidof(IDXGIResource1), reinterpret_cast<void**>(&res))) ||
res == nullptr)
{
release_shared();
return false;
}
const std::wstring name = video_share_name(m_pid);
const HRESULT hr = res->CreateSharedHandle(
nullptr, DXGI_SHARED_RESOURCE_READ | DXGI_SHARED_RESOURCE_WRITE, name.c_str(), &m_shared_handle);
res->Release();
if (FAILED(hr) || m_shared_handle == nullptr ||
FAILED(m_shared_tex->QueryInterface(__uuidof(IDXGIKeyedMutex), reinterpret_cast<void**>(&m_shared_mutex))))
{
release_shared();
return false;
}
m_share_w = w;
m_share_h = h;
return true;
}
// --- init / present / reaper -------------------------------------------------
void VkCapture::init(VkPhysicalDevice phys, VkDevice device, std::uint32_t queue_family, const Fns& fns,
unsigned long pid, std::function<void(std::uint32_t, std::uint32_t)> on_frame)
{
if (m_device != VK_NULL_HANDLE)
{
return; // already initialised
}
m_phys = phys;
m_device = device;
m_qfam = queue_family;
m_fns = fns;
m_pid = pid;
m_on_frame = std::move(on_frame);
m_stop = false;
if (!ensure_slot_pool())
{
return; // leave m_device set but the pool empty -> present() will fail format/staging checks
}
m_reaper = std::thread([this] { reaper_main(); });
}
bool VkCapture::present(VkImage image, VkFormat fmt, std::uint32_t w, std::uint32_t h, const VkSemaphore* wait,
std::uint32_t wait_count, VkSemaphore& out_sem)
{
const bool bgra = fmt == VK_FORMAT_B8G8R8A8_UNORM || fmt == VK_FORMAT_B8G8R8A8_SRGB;
const bool rgba = fmt == VK_FORMAT_R8G8B8A8_UNORM || fmt == VK_FORMAT_R8G8B8A8_SRGB;
if (m_device == VK_NULL_HANDLE || m_pool == VK_NULL_HANDLE || (!bgra && !rgba))
{
return false;
}
const double t = now_ms();
if (t - m_last_submit_ms < kMinCaptureIntervalMs)
{
return false; // throttle: mirror at ~150 Hz, not the game's (possibly 400+) present rate
}
// Pick a slot whose previous capture the reaper has finished. None free -> the reaper is behind,
// so skip this frame (the game keeps its rate; the mirror just drops a frame).
int idx = -1;
for (int n = 0; n < kSlots; ++n)
{
const int cand = (m_next + n) % kSlots;
if (!m_slots[cand].busy.load(std::memory_order_acquire))
{
idx = cand;
break;
}
}
if (idx < 0)
{
return false;
}
m_next = (idx + 1) % kSlots;
Slot& s = m_slots[idx];
if (!ensure_staging(s, w, h))
{
return false;
}
m_fns.ResetCommandBuffer(s.cmd, 0);
VkCommandBufferBeginInfo bi{VK_STRUCTURE_TYPE_COMMAND_BUFFER_BEGIN_INFO};
bi.flags = VK_COMMAND_BUFFER_USAGE_ONE_TIME_SUBMIT_BIT;
m_fns.BeginCommandBuffer(s.cmd, &bi);
auto image_barrier = [&](VkImageLayout from, VkImageLayout to, VkAccessFlags src, VkAccessFlags dst) {
VkImageMemoryBarrier b{VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER};
b.srcAccessMask = src;
b.dstAccessMask = dst;
b.oldLayout = from;
b.newLayout = to;
b.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED;
b.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED;
b.image = image;
b.subresourceRange = {VK_IMAGE_ASPECT_COLOR_BIT, 0, 1, 0, 1};
m_fns.CmdPipelineBarrier(s.cmd, VK_PIPELINE_STAGE_ALL_COMMANDS_BIT, VK_PIPELINE_STAGE_ALL_COMMANDS_BIT,
0, 0, nullptr, 0, nullptr, 1, &b);
};
image_barrier(VK_IMAGE_LAYOUT_PRESENT_SRC_KHR, VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL,
VK_ACCESS_MEMORY_READ_BIT, VK_ACCESS_TRANSFER_READ_BIT);
VkBufferImageCopy region{};
region.imageSubresource = {VK_IMAGE_ASPECT_COLOR_BIT, 0, 0, 1};
region.imageExtent = {w, h, 1};
m_fns.CmdCopyImageToBuffer(s.cmd, image, VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL, s.staging, 1, &region);
image_barrier(VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL, VK_IMAGE_LAYOUT_PRESENT_SRC_KHR,
VK_ACCESS_TRANSFER_READ_BIT, VK_ACCESS_MEMORY_READ_BIT);
m_fns.EndCommandBuffer(s.cmd);
std::vector<VkPipelineStageFlags> stages(wait_count, VK_PIPELINE_STAGE_ALL_COMMANDS_BIT);
VkSubmitInfo si{VK_STRUCTURE_TYPE_SUBMIT_INFO};
si.waitSemaphoreCount = wait_count;
si.pWaitSemaphores = wait;
si.pWaitDstStageMask = wait_count != 0 ? stages.data() : nullptr;
si.commandBufferCount = 1;
si.pCommandBuffers = &s.cmd;
si.signalSemaphoreCount = 1;
si.pSignalSemaphores = &s.present_sem;
if (m_fns.QueueSubmit(m_queue, 1, &si, s.fence) != VK_SUCCESS)
{
return false;
}
m_last_submit_ms = t;
s.w = w;
s.h = h;
s.fmt = fmt;
s.busy.store(true, std::memory_order_release);
{
std::lock_guard<std::mutex> lk(m_q_mutex);
m_pending.push_back(idx);
}
m_q_cv.notify_one();
out_sem = s.present_sem;
return true;
}
void VkCapture::reap_slot(Slot& s)
{
m_fns.WaitForFences(m_device, 1, &s.fence, VK_TRUE, UINT64_MAX);
if (!s.coherent)
{
VkMappedMemoryRange r{VK_STRUCTURE_TYPE_MAPPED_MEMORY_RANGE};
r.memory = s.mem;
r.offset = 0;
r.size = VK_WHOLE_SIZE;
m_fns.InvalidateMappedMemoryRanges(m_device, 1, &r);
}
const bool bgra = s.fmt == VK_FORMAT_B8G8R8A8_UNORM || s.fmt == VK_FORMAT_B8G8R8A8_SRGB;
const size_t row = static_cast<size_t>(s.w) * 4;
if (m_rgba.size() != row * s.h)
{
m_rgba.resize(row * s.h);
}
const auto* src = static_cast<const unsigned char*>(s.mapped);
for (std::uint32_t y = 0; y < s.h; ++y)
{
const unsigned char* in = src + static_cast<size_t>(y) * row;
unsigned char* o = m_rgba.data() + static_cast<size_t>(y) * row;
if (bgra)
{
for (std::uint32_t x = 0; x < s.w; ++x)
{
o[x * 4 + 0] = in[x * 4 + 2];
o[x * 4 + 1] = in[x * 4 + 1];
o[x * 4 + 2] = in[x * 4 + 0];
o[x * 4 + 3] = 255;
}
}
else
{
std::memcpy(o, in, row);
}
}
bool published = false;
if (ensure_d3d() && ensure_shared_texture(s.w, s.h) &&
m_shared_mutex->AcquireSync(kVideoMutexKey, 8) == S_OK)
{
m_d3d_ctx->UpdateSubresource(m_shared_tex, 0, nullptr, m_rgba.data(), static_cast<UINT>(row), 0);
m_d3d_ctx->Flush();
m_shared_mutex->ReleaseSync(kVideoMutexKey);
published = true;
}
if (published)
{
{
std::lock_guard<std::mutex> lk(m_last_mutex);
m_last = m_rgba;
m_last_w = s.w;
m_last_h = s.h;
}
m_published.fetch_add(1, std::memory_order_relaxed);
if (m_on_frame)
{
m_on_frame(s.w, s.h);
}
}
m_fns.ResetFences(m_device, 1, &s.fence);
s.busy.store(false, std::memory_order_release); // release the slot for reuse (fence already reset)
}
void VkCapture::reaper_main()
{
for (;;)
{
int idx;
{
std::unique_lock<std::mutex> lk(m_q_mutex);
m_q_cv.wait(lk, [this] { return m_stop || !m_pending.empty(); });
if (m_stop && m_pending.empty())
{
return;
}
idx = m_pending.front();
m_pending.pop_front();
}
reap_slot(m_slots[idx]);
}
}
void VkCapture::shutdown()
{
if (m_reaper.joinable())
{
{
std::lock_guard<std::mutex> lk(m_q_mutex);
m_stop = true;
}
m_q_cv.notify_all();
m_reaper.join();
}
// The reaper is gone (no more submits/reads); drain any GPU work still referencing our resources,
// then free. DeviceWaitIdle here mirrors what the inline hook / layer did before this component.
if (m_device != VK_NULL_HANDLE)
{
if (m_fns.DeviceWaitIdle != nullptr)
{
m_fns.DeviceWaitIdle(m_device);
}
free_slots();
}
release_d3d();
m_pending.clear();
m_queue = VK_NULL_HANDLE;
m_device = VK_NULL_HANDLE;
m_phys = VK_NULL_HANDLE;
}
bool VkCapture::last_frame(std::vector<unsigned char>& out, std::uint32_t& w, std::uint32_t& h)
{
std::lock_guard<std::mutex> lk(m_last_mutex);
if (m_last.empty())
{
return false;
}
out = m_last;
w = m_last_w;
h = m_last_h;
return true;
}
} // namespace coop::hook

183
hook/src/vk_capture.hpp Normal file
View File

@@ -0,0 +1,183 @@
// Shared Vulkan swap-chain read-back used by BOTH early-presence paths: the inline hook
// (hook/src/vk_hook.cpp) and the implicit layer (vk_layer/coop_vk_layer.cpp). They differ only in
// how they get into the dispatch chain; the capture itself -- copy the presented image into a shared
// keyed-mutex D3D11 texture for the host to sample -- is identical, so it lives here once.
//
// Performance contract (the whole reason this is a separate component): the read-back must NOT stall
// the game's present thread. Against a real 144 FPS game the original inline version dropped it to
// ~3 FPS because it did the GPU copy + a synchronous CPU read of the mapped staging buffer + the
// swizzle + the D3D upload all on the present thread -- and the staging buffer was HOST_COHERENT (on
// a discrete GPU that's write-combined/uncached, where a scattered CPU read runs at PCIe latency:
// ~370 ms for one 1080p frame). This component fixes both halves:
// * the present thread only records + submits the copy (sub-millisecond) and returns immediately;
// * a dedicated reaper thread waits the copy's fence, reads the staging buffer, swizzles and
// uploads -- off the critical path; and
// * the staging buffer prefers HOST_CACHED memory so the CPU read is cached, not uncached PCIe.
// A ring of in-flight slots lets capture keep pace with the game; if the reaper falls behind the
// present thread simply skips a frame (the game never waits). Proven by tests/vk_capture_perf_test.
#pragma once
#include <atomic>
#include <condition_variable>
#include <cstdint>
#include <deque>
#include <functional>
#include <mutex>
#include <thread>
#include <vector>
#include <windows.h>
#include <d3d11.h>
#include <dxgi1_2.h>
#define VK_NO_PROTOTYPES
#include <vulkan/vulkan.h>
namespace coop::hook
{
class VkCapture
{
public:
// Device entry points the read-back needs (resolved by the caller via the real
// vkGetDeviceProcAddr; GetPhysicalDeviceMemoryProperties is instance-level).
struct Fns
{
PFN_vkGetDeviceQueue GetDeviceQueue;
PFN_vkCreateCommandPool CreateCommandPool;
PFN_vkDestroyCommandPool DestroyCommandPool;
PFN_vkAllocateCommandBuffers AllocateCommandBuffers;
PFN_vkBeginCommandBuffer BeginCommandBuffer;
PFN_vkEndCommandBuffer EndCommandBuffer;
PFN_vkResetCommandBuffer ResetCommandBuffer;
PFN_vkCmdPipelineBarrier CmdPipelineBarrier;
PFN_vkCmdCopyImageToBuffer CmdCopyImageToBuffer;
PFN_vkQueueSubmit QueueSubmit;
PFN_vkCreateFence CreateFence;
PFN_vkDestroyFence DestroyFence;
PFN_vkWaitForFences WaitForFences;
PFN_vkResetFences ResetFences;
PFN_vkGetFenceStatus GetFenceStatus;
PFN_vkCreateSemaphore CreateSemaphore;
PFN_vkDestroySemaphore DestroySemaphore;
PFN_vkCreateBuffer CreateBuffer;
PFN_vkDestroyBuffer DestroyBuffer;
PFN_vkGetBufferMemoryRequirements GetBufferMemoryRequirements;
PFN_vkAllocateMemory AllocateMemory;
PFN_vkFreeMemory FreeMemory;
PFN_vkBindBufferMemory BindBufferMemory;
PFN_vkMapMemory MapMemory;
PFN_vkUnmapMemory UnmapMemory;
PFN_vkInvalidateMappedMemoryRanges InvalidateMappedMemoryRanges;
PFN_vkDeviceWaitIdle DeviceWaitIdle;
PFN_vkGetPhysicalDeviceMemoryProperties GetPhysicalDeviceMemoryProperties;
};
VkCapture() = default;
~VkCapture();
VkCapture(const VkCapture&) = delete;
VkCapture& operator=(const VkCapture&) = delete;
// Bind to the game's device + queue family and start the reaper thread. `pid` names the shared
// texture (video_share_name). `on_frame(w,h)` runs on the reaper thread after each frame is
// published (the caller does its own IPC / stat bookkeeping there). Idempotent-ish: call once.
void init(VkPhysicalDevice phys, VkDevice device, std::uint32_t queue_family, const Fns& fns,
unsigned long pid, std::function<void(std::uint32_t, std::uint32_t)> on_frame);
bool active() const { return m_device != VK_NULL_HANDLE; }
// Present-thread entry. `image` is the swap-chain image being presented (in PRESENT_SRC layout).
// If a capture was queued, returns true and sets `out_sem` to a semaphore the real present must
// wait on (our copy consumes `wait`/`wait_count` and signals `out_sem`, so present still orders
// after rendering). Returns false to present `image` unchanged (unsupported format, no free slot,
// or not initialised). Never waits on the GPU and never touches the staging buffer's bytes.
bool present(VkImage image, VkFormat fmt, std::uint32_t w, std::uint32_t h, const VkSemaphore* wait,
std::uint32_t wait_count, VkSemaphore& out_sem);
// Stop the reaper thread, drain the device (DeviceWaitIdle), and free every Vulkan/D3D resource.
// Must NOT be called from DllMain (it joins a thread). Safe if never initialised, and safe to call
// more than once.
void shutdown();
std::uint64_t frames_published() const { return m_published.load(std::memory_order_relaxed); }
// Test seam: copy the most recently published RGBA frame out (tightly packed w*4). False if none.
bool last_frame(std::vector<unsigned char>& out, std::uint32_t& w, std::uint32_t& h);
private:
static constexpr int kSlots = 4; // in-flight copies; also the present-semaphore reuse slack
struct Slot
{
VkCommandBuffer cmd = VK_NULL_HANDLE;
VkFence fence = VK_NULL_HANDLE;
VkSemaphore present_sem = VK_NULL_HANDLE;
VkBuffer staging = VK_NULL_HANDLE;
VkDeviceMemory mem = VK_NULL_HANDLE;
void* mapped = nullptr;
VkDeviceSize size = 0;
bool coherent = true; // if false, Invalidate before the CPU reads the mapping
std::uint32_t w = 0;
std::uint32_t h = 0;
VkFormat fmt = VK_FORMAT_UNDEFINED;
std::atomic<bool> busy{false}; // true from present() submit until the reaper finishes it
};
bool ensure_slot_pool();
bool ensure_staging(Slot& s, std::uint32_t w, std::uint32_t h);
void free_slots();
bool find_readback_memory(std::uint32_t type_bits, std::uint32_t& out_index, bool& out_coherent);
bool ensure_d3d();
void release_d3d();
bool ensure_shared_texture(UINT w, UINT h);
void release_shared();
void reaper_main();
void reap_slot(Slot& s);
// --- Vulkan side (present thread creates/records; reaper reads the mapping) ---
VkPhysicalDevice m_phys = VK_NULL_HANDLE;
VkDevice m_device = VK_NULL_HANDLE;
std::uint32_t m_qfam = 0;
VkQueue m_queue = VK_NULL_HANDLE;
Fns m_fns{};
VkCommandPool m_pool = VK_NULL_HANDLE;
Slot m_slots[kSlots];
int m_next = 0;
// Cap capture to ~150 Hz: a guest stream is at most the host's refresh (usually 60, at most 144),
// so mirroring every present of an uncapped 400+ FPS game is pure wasted reaper CPU. Throttling
// here -- not on the game -- keeps the mirror smooth while freeing the cores the game wants.
static constexpr double kMinCaptureIntervalMs = 1000.0 / 150.0;
double m_last_submit_ms = 0.0;
// --- D3D11 shared texture (reaper thread only) ---
ID3D11Device* m_d3d = nullptr;
ID3D11DeviceContext* m_d3d_ctx = nullptr;
ID3D11Texture2D* m_shared_tex = nullptr;
IDXGIKeyedMutex* m_shared_mutex = nullptr;
HANDLE m_shared_handle = nullptr;
UINT m_share_w = 0;
UINT m_share_h = 0;
std::vector<unsigned char> m_rgba; // reaper scratch (swizzled frame)
unsigned long m_pid = 0;
std::function<void(std::uint32_t, std::uint32_t)> m_on_frame;
// --- reaper thread + queue ---
std::thread m_reaper;
std::mutex m_q_mutex;
std::condition_variable m_q_cv;
std::deque<int> m_pending; // slot indices awaiting read-back
bool m_stop = false;
std::atomic<std::uint64_t> m_published{0};
// last published frame (for the test seam)
std::mutex m_last_mutex;
std::vector<unsigned char> m_last;
std::uint32_t m_last_w = 0;
std::uint32_t m_last_h = 0;
};
} // namespace coop::hook

View File

@@ -6,8 +6,7 @@
#include <windows.h>
#include <d3d11.h>
#include <dxgi1_2.h>
#include <dxgiformat.h>
#define VK_NO_PROTOTYPES
#define VK_USE_PLATFORM_WIN32_KHR
@@ -20,6 +19,7 @@
#include "debug_log.hpp"
#include "hook_guard.hpp"
#include "hook_registry.hpp"
#include "vk_capture.hpp"
namespace coop::hook
{
@@ -27,7 +27,7 @@ namespace coop::hook
namespace
{
DetourGate g_gate; // drains in-flight present/create detours before remove frees the Vulkan/D3D state
DetourGate g_gate; // drains in-flight present/create detours before remove frees the Vulkan state
// Capture gate. Unlike the other backends, the game caches our hk_vkQueuePresentKHR pointer at
// resolution time, so it keeps calling our detour even after the GPA hook is reset -- resetting the
// hook can't stop new detours. So removal instead closes this gate (detours then pass straight
@@ -52,55 +52,14 @@ PFN_vkGetDeviceProcAddr g_real_gdpa = nullptr;
PFN_vkCreateDevice g_real_create_device = nullptr;
PFN_vkCreateSwapchainKHR g_real_create_swapchain = nullptr;
PFN_vkQueuePresentKHR g_real_present = nullptr;
// The single tracked device's real functions (one device is the overwhelmingly common case).
struct VkFns
{
PFN_vkGetDeviceQueue GetDeviceQueue;
PFN_vkCreateCommandPool CreateCommandPool;
PFN_vkDestroyCommandPool DestroyCommandPool;
PFN_vkAllocateCommandBuffers AllocateCommandBuffers;
PFN_vkBeginCommandBuffer BeginCommandBuffer;
PFN_vkEndCommandBuffer EndCommandBuffer;
PFN_vkResetCommandBuffer ResetCommandBuffer;
PFN_vkCmdPipelineBarrier CmdPipelineBarrier;
PFN_vkCmdCopyImageToBuffer CmdCopyImageToBuffer;
PFN_vkQueueSubmit QueueSubmit;
PFN_vkCreateFence CreateFence;
PFN_vkDestroyFence DestroyFence;
PFN_vkWaitForFences WaitForFences;
PFN_vkResetFences ResetFences;
PFN_vkCreateSemaphore CreateSemaphore;
PFN_vkDestroySemaphore DestroySemaphore;
PFN_vkCreateBuffer CreateBuffer;
PFN_vkDestroyBuffer DestroyBuffer;
PFN_vkGetBufferMemoryRequirements GetBufferMemoryRequirements;
PFN_vkAllocateMemory AllocateMemory;
PFN_vkFreeMemory FreeMemory;
PFN_vkBindBufferMemory BindBufferMemory;
PFN_vkMapMemory MapMemory;
PFN_vkUnmapMemory UnmapMemory;
PFN_vkGetSwapchainImagesKHR GetSwapchainImagesKHR;
PFN_vkDeviceWaitIdle DeviceWaitIdle;
PFN_vkGetPhysicalDeviceMemoryProperties GetPhysicalDeviceMemoryProperties; // instance-level
};
VkFns g_fns{};
PFN_vkGetSwapchainImagesKHR g_get_swapchain_images = nullptr; // for our swapchain image tracking
VkInstance g_instance = VK_NULL_HANDLE;
VkPhysicalDevice g_phys = VK_NULL_HANDLE;
VkDevice g_device = VK_NULL_HANDLE;
std::uint32_t g_qfam = 0;
VkQueue g_queue = VK_NULL_HANDLE;
// Read-back resources on the game's device (created once, staging re-created on size change).
VkCommandPool g_pool = VK_NULL_HANDLE;
VkCommandBuffer g_cmd = VK_NULL_HANDLE;
VkFence g_fence = VK_NULL_HANDLE;
VkSemaphore g_present_sem = VK_NULL_HANDLE; // re-chains the present's wait (see hk_vkQueuePresentKHR)
VkBuffer g_staging = VK_NULL_HANDLE;
VkDeviceMemory g_staging_mem = VK_NULL_HANDLE;
VkDeviceSize g_staging_size = 0;
void* g_staging_mapped = nullptr;
VkCapture g_cap; // the shared, off-present-thread read-back (same component the layer uses)
// Tracked swap chains (small; engines have one or two).
struct SwapInfo
@@ -113,334 +72,12 @@ struct SwapInfo
};
std::vector<SwapInfo> g_swaps;
// Hook-owned D3D11 device + shared keyed-mutex texture (the Vulkan path has no D3D device).
ID3D11Device* g_d3d = nullptr;
ID3D11DeviceContext* g_d3d_ctx = nullptr;
ID3D11Texture2D* g_shared_tex = nullptr;
IDXGIKeyedMutex* g_shared_mutex = nullptr;
HANDLE g_shared_handle = nullptr;
UINT g_share_w = 0;
UINT g_share_h = 0;
std::vector<unsigned char> g_rgba;
// --- the real vkGetInstanceProcAddr, via the inline hook's trampoline --------
PFN_vkVoidFunction real_gipa(VkInstance inst, const char* name)
{
return g_hk_gipa.stdcall<PFN_vkVoidFunction>(inst, name);
}
// --- aux D3D11 shared texture (same contract as opengl_hook / d3d9_hook) ------
bool ensure_d3d()
{
if (g_d3d != nullptr)
{
return true;
}
const HRESULT hr =
D3D11CreateDevice(nullptr, D3D_DRIVER_TYPE_HARDWARE, nullptr, 0, nullptr, 0, D3D11_SDK_VERSION, &g_d3d,
nullptr, &g_d3d_ctx);
if (FAILED(hr) || g_d3d == nullptr)
{
logf("vk: D3D11CreateDevice failed hr=0x%08lX", static_cast<unsigned long>(hr));
return false;
}
return true;
}
void release_shared()
{
if (g_shared_mutex != nullptr)
{
g_shared_mutex->Release();
g_shared_mutex = nullptr;
}
if (g_shared_tex != nullptr)
{
g_shared_tex->Release();
g_shared_tex = nullptr;
}
if (g_shared_handle != nullptr)
{
CloseHandle(g_shared_handle);
g_shared_handle = nullptr;
}
g_share_w = g_share_h = 0;
}
bool ensure_shared_texture(UINT w, UINT h)
{
if (g_shared_tex != nullptr && g_share_w == w && g_share_h == h)
{
return true;
}
release_shared();
D3D11_TEXTURE2D_DESC desc{};
desc.Width = w;
desc.Height = h;
desc.MipLevels = 1;
desc.ArraySize = 1;
desc.Format = DXGI_FORMAT_R8G8B8A8_UNORM; // we swizzle the Vulkan image to RGBA
desc.SampleDesc.Count = 1;
desc.Usage = D3D11_USAGE_DEFAULT;
desc.BindFlags = D3D11_BIND_SHADER_RESOURCE;
desc.MiscFlags = D3D11_RESOURCE_MISC_SHARED_NTHANDLE | D3D11_RESOURCE_MISC_SHARED_KEYEDMUTEX;
if (FAILED(g_d3d->CreateTexture2D(&desc, nullptr, &g_shared_tex)) || g_shared_tex == nullptr)
{
return false;
}
IDXGIResource1* res = nullptr;
if (FAILED(g_shared_tex->QueryInterface(__uuidof(IDXGIResource1), reinterpret_cast<void**>(&res))) ||
res == nullptr)
{
release_shared();
return false;
}
const std::wstring name = video_share_name(g_pid);
const HRESULT hr = res->CreateSharedHandle(
nullptr, DXGI_SHARED_RESOURCE_READ | DXGI_SHARED_RESOURCE_WRITE, name.c_str(), &g_shared_handle);
res->Release();
if (FAILED(hr) || g_shared_handle == nullptr)
{
release_shared();
return false;
}
if (FAILED(g_shared_tex->QueryInterface(__uuidof(IDXGIKeyedMutex), reinterpret_cast<void**>(&g_shared_mutex))))
{
release_shared();
return false;
}
g_share_w = w;
g_share_h = h;
logf("vk: shared texture ready %ux%u name=%ls", w, h, name.c_str());
return true;
}
// --- Vulkan read-back resources ----------------------------------------------
bool find_host_visible_memory(std::uint32_t type_bits, std::uint32_t& out_index)
{
VkPhysicalDeviceMemoryProperties mp{};
g_fns.GetPhysicalDeviceMemoryProperties(g_phys, &mp);
for (std::uint32_t i = 0; i < mp.memoryTypeCount; ++i)
{
const bool usable = (type_bits & (1u << i)) != 0;
const bool host = (mp.memoryTypes[i].propertyFlags &
(VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT | VK_MEMORY_PROPERTY_HOST_COHERENT_BIT)) ==
(VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT | VK_MEMORY_PROPERTY_HOST_COHERENT_BIT);
if (usable && host)
{
out_index = i;
return true;
}
}
return false;
}
void release_staging()
{
if (g_staging_mapped != nullptr && g_staging_mem != VK_NULL_HANDLE)
{
g_fns.UnmapMemory(g_device, g_staging_mem);
g_staging_mapped = nullptr;
}
if (g_staging != VK_NULL_HANDLE)
{
g_fns.DestroyBuffer(g_device, g_staging, nullptr);
g_staging = VK_NULL_HANDLE;
}
if (g_staging_mem != VK_NULL_HANDLE)
{
g_fns.FreeMemory(g_device, g_staging_mem, nullptr);
g_staging_mem = VK_NULL_HANDLE;
}
g_staging_size = 0;
}
bool ensure_vk_resources(std::uint32_t w, std::uint32_t h)
{
if (g_queue == VK_NULL_HANDLE)
{
g_fns.GetDeviceQueue(g_device, g_qfam, 0, &g_queue);
}
if (g_pool == VK_NULL_HANDLE)
{
VkCommandPoolCreateInfo pci{VK_STRUCTURE_TYPE_COMMAND_POOL_CREATE_INFO};
pci.flags = VK_COMMAND_POOL_CREATE_RESET_COMMAND_BUFFER_BIT;
pci.queueFamilyIndex = g_qfam;
if (g_fns.CreateCommandPool(g_device, &pci, nullptr, &g_pool) != VK_SUCCESS)
{
return false;
}
VkCommandBufferAllocateInfo ai{VK_STRUCTURE_TYPE_COMMAND_BUFFER_ALLOCATE_INFO};
ai.commandPool = g_pool;
ai.level = VK_COMMAND_BUFFER_LEVEL_PRIMARY;
ai.commandBufferCount = 1;
if (g_fns.AllocateCommandBuffers(g_device, &ai, &g_cmd) != VK_SUCCESS)
{
return false;
}
VkFenceCreateInfo fi{VK_STRUCTURE_TYPE_FENCE_CREATE_INFO};
if (g_fns.CreateFence(g_device, &fi, nullptr, &g_fence) != VK_SUCCESS)
{
return false;
}
VkSemaphoreCreateInfo si{VK_STRUCTURE_TYPE_SEMAPHORE_CREATE_INFO};
if (g_fns.CreateSemaphore(g_device, &si, nullptr, &g_present_sem) != VK_SUCCESS)
{
return false;
}
}
const VkDeviceSize need = static_cast<VkDeviceSize>(w) * h * 4;
if (g_staging != VK_NULL_HANDLE && g_staging_size == need)
{
return true;
}
release_staging();
VkBufferCreateInfo bci{VK_STRUCTURE_TYPE_BUFFER_CREATE_INFO};
bci.size = need;
bci.usage = VK_BUFFER_USAGE_TRANSFER_DST_BIT;
bci.sharingMode = VK_SHARING_MODE_EXCLUSIVE;
if (g_fns.CreateBuffer(g_device, &bci, nullptr, &g_staging) != VK_SUCCESS)
{
return false;
}
VkMemoryRequirements mr{};
g_fns.GetBufferMemoryRequirements(g_device, g_staging, &mr);
std::uint32_t mem_type = 0;
if (!find_host_visible_memory(mr.memoryTypeBits, mem_type))
{
release_staging();
return false;
}
VkMemoryAllocateInfo mai{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO};
mai.allocationSize = mr.size;
mai.memoryTypeIndex = mem_type;
if (g_fns.AllocateMemory(g_device, &mai, nullptr, &g_staging_mem) != VK_SUCCESS ||
g_fns.BindBufferMemory(g_device, g_staging, g_staging_mem, 0) != VK_SUCCESS ||
g_fns.MapMemory(g_device, g_staging_mem, 0, VK_WHOLE_SIZE, 0, &g_staging_mapped) != VK_SUCCESS)
{
release_staging();
return false;
}
g_staging_size = need;
return true;
}
void image_barrier(VkCommandBuffer cb, VkImage img, VkImageLayout from, VkImageLayout to, VkAccessFlags src,
VkAccessFlags dst)
{
VkImageMemoryBarrier b{VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER};
b.srcAccessMask = src;
b.dstAccessMask = dst;
b.oldLayout = from;
b.newLayout = to;
b.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED;
b.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED;
b.image = img;
b.subresourceRange = {VK_IMAGE_ASPECT_COLOR_BIT, 0, 1, 0, 1};
g_fns.CmdPipelineBarrier(cb, VK_PIPELINE_STAGE_ALL_COMMANDS_BIT, VK_PIPELINE_STAGE_ALL_COMMANDS_BIT, 0, 0,
nullptr, 0, nullptr, 1, &b);
}
// Read `image` (in PRESENT_SRC layout) back into the shared texture. `wait`/`wait_count` are the
// present's wait semaphores, which our submit consumes and replaces with g_present_sem (returned
// via *out_sem) so the real present orders correctly after our copy without double-waiting.
bool capture_present(VkImage image, VkFormat fmt, std::uint32_t w, std::uint32_t h, const VkSemaphore* wait,
std::uint32_t wait_count, VkSemaphore& out_sem)
{
const bool bgra = fmt == VK_FORMAT_B8G8R8A8_UNORM || fmt == VK_FORMAT_B8G8R8A8_SRGB;
const bool rgba = fmt == VK_FORMAT_R8G8B8A8_UNORM || fmt == VK_FORMAT_R8G8B8A8_SRGB;
if (!bgra && !rgba)
{
if (!g_unsupported_logged)
{
logf("vk: unsupported swapchain format=%d (only 8888 BGRA/RGBA); idle", static_cast<int>(fmt));
g_unsupported_logged = true;
}
return false;
}
if (!ensure_d3d() || !ensure_shared_texture(w, h) || !ensure_vk_resources(w, h))
{
return false;
}
g_fns.ResetCommandBuffer(g_cmd, 0);
VkCommandBufferBeginInfo bi{VK_STRUCTURE_TYPE_COMMAND_BUFFER_BEGIN_INFO};
bi.flags = VK_COMMAND_BUFFER_USAGE_ONE_TIME_SUBMIT_BIT;
g_fns.BeginCommandBuffer(g_cmd, &bi);
image_barrier(g_cmd, image, VK_IMAGE_LAYOUT_PRESENT_SRC_KHR, VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL,
VK_ACCESS_MEMORY_READ_BIT, VK_ACCESS_TRANSFER_READ_BIT);
VkBufferImageCopy region{};
region.imageSubresource = {VK_IMAGE_ASPECT_COLOR_BIT, 0, 0, 1};
region.imageExtent = {w, h, 1};
g_fns.CmdCopyImageToBuffer(g_cmd, image, VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL, g_staging, 1, &region);
image_barrier(g_cmd, image, VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL, VK_IMAGE_LAYOUT_PRESENT_SRC_KHR,
VK_ACCESS_TRANSFER_READ_BIT, VK_ACCESS_MEMORY_READ_BIT);
g_fns.EndCommandBuffer(g_cmd);
std::vector<VkPipelineStageFlags> wait_stages(wait_count, VK_PIPELINE_STAGE_ALL_COMMANDS_BIT);
VkSubmitInfo si{VK_STRUCTURE_TYPE_SUBMIT_INFO};
si.waitSemaphoreCount = wait_count;
si.pWaitSemaphores = wait;
si.pWaitDstStageMask = wait_count != 0 ? wait_stages.data() : nullptr;
si.commandBufferCount = 1;
si.pCommandBuffers = &g_cmd;
si.signalSemaphoreCount = 1;
si.pSignalSemaphores = &g_present_sem;
g_fns.ResetFences(g_device, 1, &g_fence);
if (g_fns.QueueSubmit(g_queue, 1, &si, g_fence) != VK_SUCCESS)
{
return false;
}
g_fns.WaitForFences(g_device, 1, &g_fence, VK_TRUE, UINT64_MAX);
// Swizzle the mapped staging (tightly packed w*4) into RGBA and upload.
const size_t row = static_cast<size_t>(w) * 4;
if (g_rgba.size() != row * h)
{
g_rgba.resize(row * h);
}
const auto* src = static_cast<const unsigned char*>(g_staging_mapped);
for (std::uint32_t y = 0; y < h; ++y)
{
const unsigned char* s = src + static_cast<size_t>(y) * row;
unsigned char* o = g_rgba.data() + static_cast<size_t>(y) * row;
if (bgra)
{
for (std::uint32_t x = 0; x < w; ++x)
{
o[x * 4 + 0] = s[x * 4 + 2];
o[x * 4 + 1] = s[x * 4 + 1];
o[x * 4 + 2] = s[x * 4 + 0];
o[x * 4 + 3] = 255;
}
}
else
{
std::memcpy(o, s, row);
}
}
bool shared = false;
if (g_shared_mutex->AcquireSync(kVideoMutexKey, 8) == S_OK)
{
g_d3d_ctx->UpdateSubresource(g_shared_tex, 0, nullptr, g_rgba.data(), static_cast<UINT>(row), 0);
g_d3d_ctx->Flush();
g_shared_mutex->ReleaseSync(kVideoMutexKey);
shared = true;
}
out_sem = g_present_sem;
if (shared)
{
g_present_captured.store(true, std::memory_order_relaxed);
g_frames_shared.fetch_add(1, std::memory_order_relaxed);
if (g_ipc != nullptr)
{
g_ipc->publish_video_frame(w, h, static_cast<std::uint32_t>(DXGI_FORMAT_R8G8B8A8_UNORM));
}
}
return true;
}
const SwapInfo* find_swap(VkSwapchainKHR sc)
{
for (const SwapInfo& s : g_swaps)
@@ -473,8 +110,8 @@ VKAPI_ATTR VkResult VKAPI_CALL hk_vkQueuePresentKHR(VkQueue queue, const VkPrese
if (s != nullptr && pPresentInfo->pImageIndices[0] < s->images.size())
{
VkSemaphore chained = VK_NULL_HANDLE;
if (capture_present(s->images[pPresentInfo->pImageIndices[0]], s->fmt, s->w, s->h,
pPresentInfo->pWaitSemaphores, pPresentInfo->waitSemaphoreCount, chained))
if (g_cap.present(s->images[pPresentInfo->pImageIndices[0]], s->fmt, s->w, s->h,
pPresentInfo->pWaitSemaphores, pPresentInfo->waitSemaphoreCount, chained))
{
// Replace the present's wait with our chained semaphore (our submit consumed the
// originals and signals this one), so the present still orders after rendering.
@@ -493,7 +130,7 @@ VKAPI_ATTR VkResult VKAPI_CALL hk_vkCreateSwapchainKHR(VkDevice device, const Vk
{
DetourGate::Guard guard(g_gate); // keep g_swaps stable while remove may be clearing it
const VkResult r = g_real_create_swapchain(device, ci, alloc, out);
if (r == VK_SUCCESS && out != nullptr && g_fns.GetSwapchainImagesKHR != nullptr)
if (r == VK_SUCCESS && out != nullptr && g_get_swapchain_images != nullptr)
{
SwapInfo info{};
info.sc = *out;
@@ -501,9 +138,9 @@ VKAPI_ATTR VkResult VKAPI_CALL hk_vkCreateSwapchainKHR(VkDevice device, const Vk
info.w = ci->imageExtent.width;
info.h = ci->imageExtent.height;
std::uint32_t n = 0;
g_fns.GetSwapchainImagesKHR(device, *out, &n, nullptr);
g_get_swapchain_images(device, *out, &n, nullptr);
info.images.resize(n);
g_fns.GetSwapchainImagesKHR(device, *out, &n, info.images.data());
g_get_swapchain_images(device, *out, &n, info.images.data());
g_swaps.push_back(std::move(info));
// (Don't log the VkSwapchainKHR handle: it's a uint64_t on x86, not a pointer.)
logf("vk: swapchain %ux%u fmt=%d images=%u", ci->imageExtent.width, ci->imageExtent.height,
@@ -512,10 +149,11 @@ VKAPI_ATTR VkResult VKAPI_CALL hk_vkCreateSwapchainKHR(VkDevice device, const Vk
return r;
}
// Resolve every device function we need for the read-back via the real vkGetDeviceProcAddr.
void load_device_fns(VkDevice device)
// Resolve every device function VkCapture needs (via the real vkGetDeviceProcAddr) and start it.
void start_capture(VkDevice device)
{
#define LOAD(field, vkname) g_fns.field = reinterpret_cast<PFN_##vkname>(g_real_gdpa(device, #vkname))
VkCapture::Fns f{};
#define LOAD(field, vkname) f.field = reinterpret_cast<PFN_##vkname>(g_real_gdpa(device, #vkname))
LOAD(GetDeviceQueue, vkGetDeviceQueue);
LOAD(CreateCommandPool, vkCreateCommandPool);
LOAD(DestroyCommandPool, vkDestroyCommandPool);
@@ -530,6 +168,7 @@ void load_device_fns(VkDevice device)
LOAD(DestroyFence, vkDestroyFence);
LOAD(WaitForFences, vkWaitForFences);
LOAD(ResetFences, vkResetFences);
LOAD(GetFenceStatus, vkGetFenceStatus);
LOAD(CreateSemaphore, vkCreateSemaphore);
LOAD(DestroySemaphore, vkDestroySemaphore);
LOAD(CreateBuffer, vkCreateBuffer);
@@ -540,11 +179,23 @@ void load_device_fns(VkDevice device)
LOAD(BindBufferMemory, vkBindBufferMemory);
LOAD(MapMemory, vkMapMemory);
LOAD(UnmapMemory, vkUnmapMemory);
LOAD(GetSwapchainImagesKHR, vkGetSwapchainImagesKHR);
LOAD(InvalidateMappedMemoryRanges, vkInvalidateMappedMemoryRanges);
LOAD(DeviceWaitIdle, vkDeviceWaitIdle);
#undef LOAD
g_fns.GetPhysicalDeviceMemoryProperties =
reinterpret_cast<PFN_vkGetPhysicalDeviceMemoryProperties>(real_gipa(g_instance, "vkGetPhysicalDeviceMemoryProperties"));
f.GetPhysicalDeviceMemoryProperties = reinterpret_cast<PFN_vkGetPhysicalDeviceMemoryProperties>(
real_gipa(g_instance, "vkGetPhysicalDeviceMemoryProperties"));
g_get_swapchain_images =
reinterpret_cast<PFN_vkGetSwapchainImagesKHR>(g_real_gdpa(device, "vkGetSwapchainImagesKHR"));
g_cap.init(g_phys, device, g_qfam, f, g_pid, [](std::uint32_t w, std::uint32_t h) {
// Reaper thread, after each frame is mirrored into the shared texture.
g_present_captured.store(true, std::memory_order_relaxed);
g_frames_shared.fetch_add(1, std::memory_order_relaxed);
if (g_ipc != nullptr)
{
g_ipc->publish_video_frame(w, h, static_cast<std::uint32_t>(DXGI_FORMAT_R8G8B8A8_UNORM));
}
});
}
VKAPI_ATTR VkResult VKAPI_CALL hk_vkCreateDevice(VkPhysicalDevice phys, const VkDeviceCreateInfo* ci,
@@ -561,8 +212,8 @@ VKAPI_ATTR VkResult VKAPI_CALL hk_vkCreateDevice(VkPhysicalDevice phys, const Vk
g_real_create_swapchain =
reinterpret_cast<PFN_vkCreateSwapchainKHR>(g_real_gdpa(*out, "vkCreateSwapchainKHR"));
g_real_present = reinterpret_cast<PFN_vkQueuePresentKHR>(g_real_gdpa(*out, "vkQueuePresentKHR"));
load_device_fns(*out);
// Arm capture only once every real_* pointer + g_fns is populated (release pairs with the
start_capture(*out);
// Arm capture only once every real_* pointer + VkCapture is populated (release pairs with the
// present detour's acquire load of the gate, so it sees a fully-initialised state).
g_capture_enabled.store(true, std::memory_order_release);
logf("vk: device created (qfam=%u) -- present capture armed", g_qfam);
@@ -666,49 +317,17 @@ void remove_vk_hooks()
{
// Close the capture gate and reset the GPA hook FIRST (so no new detour captures and future
// resolutions aren't intercepted), then drain any present/create detour still in-flight on the
// game's thread BEFORE freeing the read-back resources it reads. The game keeps calling our
// cached present detour, but with the gate closed it now passes straight through to the real
// present without touching freed state.
// game's thread BEFORE shutting the capture down. The game keeps calling our cached present
// detour, but with the gate closed it now passes straight through to the real present.
g_capture_enabled.store(false, std::memory_order_release);
g_hk_gipa = {};
g_gate.drain();
if (g_device != VK_NULL_HANDLE && g_fns.DeviceWaitIdle != nullptr)
{
g_fns.DeviceWaitIdle(g_device);
release_staging();
if (g_present_sem != VK_NULL_HANDLE)
{
g_fns.DestroySemaphore(g_device, g_present_sem, nullptr);
g_present_sem = VK_NULL_HANDLE;
}
if (g_fence != VK_NULL_HANDLE)
{
g_fns.DestroyFence(g_device, g_fence, nullptr);
g_fence = VK_NULL_HANDLE;
}
if (g_pool != VK_NULL_HANDLE)
{
g_fns.DestroyCommandPool(g_device, g_pool, nullptr);
g_pool = VK_NULL_HANDLE;
}
}
g_cap.shutdown(); // joins the reaper, drains the device, frees the read-back resources
hook_set_installed(g_id_present, false);
release_shared();
if (g_d3d_ctx != nullptr)
{
g_d3d_ctx->Release();
g_d3d_ctx = nullptr;
}
if (g_d3d != nullptr)
{
g_d3d->Release();
g_d3d = nullptr;
}
g_swaps.clear();
g_rgba.clear();
g_cmd = VK_NULL_HANDLE;
g_queue = VK_NULL_HANDLE;
g_get_swapchain_images = nullptr;
g_device = VK_NULL_HANDLE;
g_instance = VK_NULL_HANDLE;
g_real_gdpa = nullptr;