mirror of
https://github.com/spice2x/spice2x.github.io.git
synced 2026-10-02 08:18:14 -07:00
21e7d24ed3
## Link to GitHub Issue or related Pull Request, if one exists #0 ## Description of change Capturing a frame for the API stream made the game wait for `GetRenderTargetData` in the middle of its present, roughly 1270us per frame at 1080p. A 120Hz cab visibly lost frames for as long as a viewer was connected. The present thread now only issues a `StretchRect` into a render target we own, which is queued rather than waited on, and a pool thread does the readback and the pixel conversion. That takes the present thread cost to 1-4us. Each snapshot is read on the request after the one that took it, so the blit and its transfer have a full frame to land and the read does not stall on the GPU either, at the cost of one frame of stream latency. Only streaming takes this path, and only on a device created with `D3DCREATE_MULTITHREADED`. Screenshots, `capture.get_jpg` and the `THREAD_BAN` models keep the existing inline readback unchanged. Also raises the x264 encoder from `i_threads = 1` to 4, which was holding a 1080p60 stream to 41fps and making a keyframe cost 12.7ms against 6.6ms for an ordinary frame. Capped rather than automatic because this encodes on the same machine it is capturing. ## Testing tested against iidx33, which was the most sensitive to frame drops
91 lines
3.5 KiB
C++
91 lines
3.5 KiB
C++
#pragma once
|
|
|
|
#include <cstdint>
|
|
#include <memory>
|
|
#include <optional>
|
|
|
|
#include <d3d9.h>
|
|
|
|
namespace d3d9_readback {
|
|
|
|
struct SurfaceReleaser {
|
|
void operator()(IDirect3DSurface9 *surface) const {
|
|
surface->Release();
|
|
}
|
|
};
|
|
|
|
using SurfacePtr = std::unique_ptr<IDirect3DSurface9, SurfaceReleaser>;
|
|
|
|
// system memory copy of a back buffer; locking it neither stalls the GPU nor reads over PCIe
|
|
struct BackbufferCopy {
|
|
int screen {};
|
|
D3DSURFACE_DESC desc {};
|
|
IDirect3DDevice9 *device = nullptr;
|
|
SurfacePtr surface;
|
|
bool pooled = false;
|
|
|
|
BackbufferCopy() = default;
|
|
BackbufferCopy(BackbufferCopy &&) noexcept = default;
|
|
BackbufferCopy &operator=(BackbufferCopy &&) noexcept = default;
|
|
BackbufferCopy(const BackbufferCopy &) = delete;
|
|
BackbufferCopy &operator=(const BackbufferCopy &) = delete;
|
|
~BackbufferCopy();
|
|
};
|
|
|
|
// pooled copies reuse surfaces across calls and return them once the copy is destroyed,
|
|
// so the caller must keep it alive for as long as the pixels are being read
|
|
std::optional<BackbufferCopy> acquire_backbuffer_copy(
|
|
IDirect3DDevice9 *device,
|
|
IDirect3DSwapChain9 *swap_chain,
|
|
int screen,
|
|
bool pooled);
|
|
|
|
// GPU side copy of a back buffer, taken while the contents are still the frame that was
|
|
// presented, so that reading them into system memory no longer has to happen before it
|
|
struct Snapshot {
|
|
int screen {};
|
|
D3DSURFACE_DESC desc {};
|
|
IDirect3DDevice9 *device = nullptr;
|
|
SurfacePtr surface;
|
|
uint64_t generation {};
|
|
|
|
// when the blit was issued, so a frame left behind by a break in the request stream can
|
|
// be recognised as stale rather than handed over
|
|
uint64_t issued_us {};
|
|
|
|
Snapshot() = default;
|
|
Snapshot(Snapshot &&) noexcept = default;
|
|
Snapshot &operator=(Snapshot &&) noexcept = default;
|
|
Snapshot(const Snapshot &) = delete;
|
|
Snapshot &operator=(const Snapshot &) = delete;
|
|
~Snapshot();
|
|
};
|
|
|
|
// for the present thread, between the last EndScene and Present. blits the current frame,
|
|
// then returns the snapshot taken on the *previous* call: waiting a frame before reading
|
|
// means the blit and its system memory transfer have already happened, so the read does not
|
|
// stall on the GPU. costs the stream one frame of latency.
|
|
//
|
|
// returns nothing on the first call of a stream, and whenever the frame could not be taken,
|
|
// which is the caller's cue to skip rather than to wait
|
|
std::optional<Snapshot> snapshot_backbuffer(
|
|
IDirect3DDevice9 *device,
|
|
IDirect3DSwapChain9 *swap_chain,
|
|
int screen);
|
|
|
|
// the expensive half, for a thread that is not the present thread. only legal on a device
|
|
// created with D3DCREATE_MULTITHREADED
|
|
std::optional<BackbufferCopy> read_snapshot(Snapshot snapshot);
|
|
|
|
// false once a device has refused to give up a render target matching its back buffer,
|
|
// which leaves reading the back buffer directly as the only way to capture it
|
|
bool snapshots_supported();
|
|
|
|
// snapshot targets live in the default pool, so unlike the readback surfaces they have to
|
|
// be gone before a Reset and not merely before the device is released
|
|
void discard_snapshot_targets(IDirect3DDevice9 *device);
|
|
|
|
// pooled surfaces hold references on the device; call this before releasing it
|
|
void release_device_resources(IDirect3DDevice9 *device);
|
|
}
|