File size: 8,405 Bytes
5506acb | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 | // perf.cpp — measures the resolution-scaling performance benefit of the SVR
// upscale path, isolated with GPU timestamp queries. No VR runtime involved.
//
// The saving from any upscaler is that the game renders fewer, cheaper pixels;
// the kernel then reconstructs full resolution at a small fixed cost. This
// harness renders an expensive procedural fragment shader (domain-warped fBm,
// octave count = per-pixel shading load) to eye-sized render targets, times
// the scene on the GPU at full resolution and at renderScale R, runs the real
// SVR kernel (svr_cuda.dll) to reconstruct R -> full, and reports the net
// frame speedup against the pixel-count ceiling. The octave sweep spans light
// to heavy shading so the speedup curve is visible; a real title sits on it
// wherever its own per-pixel cost lands.
#include <d3d11.h>
#include <d3dcompiler.h>
#include <cstdint>
#include <cstdio>
#include <vector>
#pragma comment(lib, "d3d11.lib")
#pragma comment(lib, "d3dcompiler.lib")
static const char* kShader = R"(
cbuffer CB : register(b0) { float4 p; } // x=octaves, yz=size, w=time
struct VSOut { float4 pos : SV_Position; float2 uv : TEXCOORD0; };
VSOut vs(uint id : SV_VertexID) {
VSOut o; float2 xy = float2((id << 1) & 2, id & 2);
o.pos = float4(xy * 2.0 - 1.0, 0.0, 1.0); o.uv = xy; return o;
}
float hash(float2 n){ return frac(sin(dot(n, float2(12.9898,78.233)))*43758.5453); }
float vnoise(float2 x){
float2 i=floor(x), f=frac(x); f=f*f*(3.0-2.0*f);
float a=hash(i), b=hash(i+float2(1,0)), c=hash(i+float2(0,1)), d=hash(i+float2(1,1));
return lerp(lerp(a,b,f.x), lerp(c,d,f.x), f.y);
}
float4 ps(VSOut v) : SV_Target {
float2 uv = v.uv * p.yz * 0.02;
float2 q = uv; float acc = 0.0, amp = 0.5;
int N = (int)p.x;
[loop] for (int i = 0; i < N; ++i) {
q += float2(vnoise(q*1.7 + p.w), vnoise(q*1.7 - p.w)) * 0.35;
acc += amp * vnoise(q); amp *= 0.6; q *= 1.9;
}
float3 col = 0.5 + 0.5*cos(6.2831*(acc + float3(0.0,0.33,0.67)));
return float4(col, 1.0);
}
)";
typedef int (*SvrProcessFn)(ID3D11Texture2D*, int, int, int, int, int,
ID3D11Texture2D*, int, int, int, int, int,
float, float, float, float, float);
static ID3D11Texture2D* mkrt(ID3D11Device* d, int w, int h) {
D3D11_TEXTURE2D_DESC td = {};
td.Width = w; td.Height = h; td.MipLevels = 1; td.ArraySize = 1;
td.Format = DXGI_FORMAT_R8G8B8A8_UNORM; td.SampleDesc.Count = 1;
td.Usage = D3D11_USAGE_DEFAULT;
td.BindFlags = D3D11_BIND_RENDER_TARGET | D3D11_BIND_SHADER_RESOURCE;
ID3D11Texture2D* t = nullptr; d->CreateTexture2D(&td, nullptr, &t); return t;
}
int main() {
HMODULE svr = LoadLibraryW(L"svr_cuda.dll");
SvrProcessFn svr_process = svr ? (SvrProcessFn)GetProcAddress(svr, "svr_process") : nullptr;
if (!svr_process) { printf("svr_cuda.dll svr_process missing\n"); return 1; }
ID3D11Device* dev = nullptr; ID3D11DeviceContext* ctx = nullptr;
D3D_FEATURE_LEVEL fl;
if (FAILED(D3D11CreateDevice(nullptr, D3D_DRIVER_TYPE_HARDWARE, nullptr, 0,
nullptr, 0, D3D11_SDK_VERSION, &dev, &fl, &ctx))) {
printf("device create failed\n"); return 1;
}
ID3DBlob *vsb=nullptr,*psb=nullptr,*eb=nullptr;
D3DCompile(kShader, strlen(kShader), 0,0,0, "vs","vs_5_0",0,0,&vsb,&eb);
if (FAILED(D3DCompile(kShader, strlen(kShader), 0,0,0, "ps","ps_5_0",0,0,&psb,&eb))) {
printf("ps: %s\n", eb?(char*)eb->GetBufferPointer():"?"); return 1; }
ID3D11VertexShader* vs=nullptr; ID3D11PixelShader* ps=nullptr;
dev->CreateVertexShader(vsb->GetBufferPointer(), vsb->GetBufferSize(),0,&vs);
dev->CreatePixelShader(psb->GetBufferPointer(), psb->GetBufferSize(),0,&ps);
D3D11_BUFFER_DESC bd={}; bd.ByteWidth=16; bd.Usage=D3D11_USAGE_DEFAULT;
bd.BindFlags=D3D11_BIND_CONSTANT_BUFFER; ID3D11Buffer* cb=nullptr;
dev->CreateBuffer(&bd,nullptr,&cb);
D3D11_RASTERIZER_DESC rd={}; rd.FillMode=D3D11_FILL_SOLID; rd.CullMode=D3D11_CULL_NONE;
ID3D11RasterizerState* rs=nullptr; dev->CreateRasterizerState(&rd,&rs);
ctx->RSSetState(rs);
ID3D11Query *qd=nullptr,*q0=nullptr,*q1=nullptr;
D3D11_QUERY_DESC dqd={D3D11_QUERY_TIMESTAMP_DISJOINT,0}; dev->CreateQuery(&dqd,&qd);
D3D11_QUERY_DESC tqd={D3D11_QUERY_TIMESTAMP,0};
dev->CreateQuery(&tqd,&q0); dev->CreateQuery(&tqd,&q1);
auto render_two_eyes = [&](ID3D11RenderTargetView* rL, ID3D11RenderTargetView* rR,
int w, int h, int oct, float t) {
float p[4] = {(float)oct, (float)w, (float)h, t};
ctx->UpdateSubresource(cb,0,nullptr,p,0,0);
ctx->IASetPrimitiveTopology(D3D11_PRIMITIVE_TOPOLOGY_TRIANGLELIST);
ctx->IASetInputLayout(nullptr);
ctx->VSSetShader(vs,nullptr,0); ctx->VSSetConstantBuffers(0,1,&cb);
ctx->PSSetShader(ps,nullptr,0); ctx->PSSetConstantBuffers(0,1,&cb);
ctx->RSSetState(rs);
D3D11_VIEWPORT vp={0,0,(float)w,(float)h,0,1}; ctx->RSSetViewports(1,&vp);
ID3D11RenderTargetView* rts[2]={rL,rR};
for (int e=0;e<2;++e){ ctx->OMSetRenderTargets(1,&rts[e],nullptr); ctx->Draw(3,0); }
};
auto gpu_ms = [&](auto&& fn)->double{
// warm
for(int i=0;i<3;++i) fn(); ctx->Flush();
double best=1e30;
for(int rep=0;rep<12;++rep){
ctx->Begin(qd); ctx->End(q0); fn(); ctx->End(q1); ctx->End(qd);
ctx->Flush();
D3D11_QUERY_DATA_TIMESTAMP_DISJOINT dj; UINT64 t0,t1;
while(ctx->GetData(qd,&dj,sizeof(dj),0)!=S_OK){}
while(ctx->GetData(q0,&t0,sizeof(t0),0)!=S_OK){}
while(ctx->GetData(q1,&t1,sizeof(t1),0)!=S_OK){}
if(dj.Disjoint) continue;
double ms = 1000.0*(double)(t1-t0)/(double)dj.Frequency;
if(ms<best) best=ms;
}
return best;
};
auto kernel_ms = [&](ID3D11Texture2D* srcL, ID3D11Texture2D* srcR, int iw,int ih,
ID3D11Texture2D* dstL, ID3D11Texture2D* dstR, int ow,int oh)->double{
float r1=0.65f*oh, r0=0.55f*r1;
auto once=[&]{ svr_process(srcL,0,0,0,iw,ih,dstL,0,0,0,ow,oh,0.46f*ow,0.5f*oh,r0,r1,0.4f);
svr_process(srcR,0,0,0,iw,ih,dstR,0,0,0,ow,oh,0.54f*ow,0.5f*oh,r0,r1,0.4f); };
for(int i=0;i<5;++i) once(); // warm + register
LARGE_INTEGER f,a,b; QueryPerformanceFrequency(&f);
QueryPerformanceCounter(&a);
const int N=60; for(int i=0;i<N;++i) once();
QueryPerformanceCounter(&b);
return 1000.0*(double)(b.QuadPart-a.QuadPart)/(double)f.QuadPart/N;
};
struct Res { int w,h; const char* name; };
Res shapes[] = { {1440,1600,"Quest3 low"}, {2016,2240,"Quest3 med"},
{2464,2736,"Quest3 high"} };
float scales[] = { 0.90f, 0.77f, 0.67f, 0.59f };
int octaves[] = { 4, 8, 16, 24 };
printf("SVR resolution-scaling benefit (GPU timestamps, per frame = 2 eyes)\n");
printf("shading load = fBm octave count; a real title sits where its per-pixel cost lands\n\n");
for (auto& sh : shapes) {
ID3D11Texture2D* fullL=mkrt(dev,sh.w,sh.h); ID3D11Texture2D* fullR=mkrt(dev,sh.w,sh.h);
ID3D11RenderTargetView *fLrtv=nullptr,*fRrtv=nullptr;
dev->CreateRenderTargetView(fullL,nullptr,&fLrtv);
dev->CreateRenderTargetView(fullR,nullptr,&fRrtv);
printf("== %s %dx%d/eye ==\n", sh.name, sh.w, sh.h);
printf("%7s %8s | %9s %9s %8s %8s | %8s %8s\n",
"scale","pix%","full ms","scaled ms","kern ms","frame ms","speedup","ceiling");
for (float sc : scales) {
int rw=(int)(sh.w*sc+0.5f), rh=(int)(sh.h*sc+0.5f);
ID3D11Texture2D* redL=mkrt(dev,rw,rh); ID3D11Texture2D* redR=mkrt(dev,rw,rh);
ID3D11RenderTargetView *rLrtv=nullptr,*rRrtv=nullptr;
dev->CreateRenderTargetView(redL,nullptr,&rLrtv);
dev->CreateRenderTargetView(redR,nullptr,&rRrtv);
double pixr = (double)(rw*rh)/(double)(sh.w*sh.h);
for (int oct : octaves) {
double tf = gpu_ms([&]{ render_two_eyes(fLrtv,fRrtv,sh.w,sh.h,oct,1.3f); });
double ts = gpu_ms([&]{ render_two_eyes(rLrtv,rRrtv,rw,rh,oct,1.3f); });
double tk = kernel_ms(redL,redR,rw,rh,fullL,fullR,sh.w,sh.h);
double frame = ts + tk;
printf("%6.2f %6.1f%% | %9.3f %9.3f %8.3f %8.3f | %7.2fx %7.2fx (oct %d)\n",
sc, 100.0*pixr, tf, ts, tk, frame, tf/frame, 1.0/pixr, oct);
}
rLrtv->Release(); rRrtv->Release(); redL->Release(); redR->Release();
}
fLrtv->Release(); fRrtv->Release(); fullL->Release(); fullR->Release();
printf("\n");
}
printf("done\n");
return 0;
}
|