mirror of
https://github.com/gnif/LookingGlass.git
synced 2026-08-09 00:31:31 +00:00
[idd] rgb24: compare stable candidate timings
Ignore allocation alignment when matching retained candidate resources. D3D12 may resolve automatic alignment requests in GetDesc. Comparing that value recreated the candidate for every benchmark sample. Measure native and packed over matching wall-clock intervals. Use preparation and publication intervals, excluding the cadence hold. This avoids mixing GPU timestamps with fallback timing. The CPU IVSHMEM copy is included only on indirect systems.
This commit is contained in:
@@ -598,6 +598,7 @@ void CSwapChainProcessor::CompletionFunction(
|
|||||||
uint64_t prepareReady;
|
uint64_t prepareReady;
|
||||||
uint64_t prepareGPUStart;
|
uint64_t prepareGPUStart;
|
||||||
uint64_t prepareGPUEnd;
|
uint64_t prepareGPUEnd;
|
||||||
|
uint64_t timingStart;
|
||||||
bool prepareTimingValid;
|
bool prepareTimingValid;
|
||||||
AcquireSRWLockShared(&sc->m_candidateLock);
|
AcquireSRWLockShared(&sc->m_candidateLock);
|
||||||
const FrameCandidate& candidate = sc->m_candidates[candidateIndex];
|
const FrameCandidate& candidate = sc->m_candidates[candidateIndex];
|
||||||
@@ -605,6 +606,7 @@ void CSwapChainProcessor::CompletionFunction(
|
|||||||
prepareReady = candidate.prepareReady;
|
prepareReady = candidate.prepareReady;
|
||||||
prepareGPUStart = candidate.prepareGPUStart;
|
prepareGPUStart = candidate.prepareGPUStart;
|
||||||
prepareGPUEnd = candidate.prepareGPUEnd;
|
prepareGPUEnd = candidate.prepareGPUEnd;
|
||||||
|
timingStart = candidate.timingStart;
|
||||||
prepareTimingValid = candidate.prepareTimingValid;
|
prepareTimingValid = candidate.prepareTimingValid;
|
||||||
ReleaseSRWLockShared(&sc->m_candidateLock);
|
ReleaseSRWLockShared(&sc->m_candidateLock);
|
||||||
|
|
||||||
@@ -614,8 +616,8 @@ void CSwapChainProcessor::CompletionFunction(
|
|||||||
uint64_t indirectCopyTime = 0;
|
uint64_t indirectCopyTime = 0;
|
||||||
if (sc->m_dx12Device->IsIndirectCopy())
|
if (sc->m_dx12Device->IsIndirectCopy())
|
||||||
{
|
{
|
||||||
// GPU timestamps end at the readback copy. Include the following CPU copy
|
// GPU timestamps end at the readback copy. Track the following CPU copy
|
||||||
// to IVSHMEM in effect benchmark samples without charging it to Ready.
|
// separately for frame metrics; benchmark wall time includes it directly.
|
||||||
const uint64_t indirectCopyStart = CFrameScheduler::Nanotime();
|
const uint64_t indirectCopyStart = CFrameScheduler::Nanotime();
|
||||||
sc->m_devContext->WriteFrameBuffer(
|
sc->m_devContext->WriteFrameBuffer(
|
||||||
fbRes->GetFrameIndex(), fbRes->GetMap(), 0, fbRes->GetFrameSize(), false);
|
fbRes->GetFrameIndex(), fbRes->GetMap(), 0, fbRes->GetFrameSize(), false);
|
||||||
@@ -652,9 +654,18 @@ void CSwapChainProcessor::CompletionFunction(
|
|||||||
const uint64_t measured = postProcessTime + copyTime;
|
const uint64_t measured = postProcessTime + copyTime;
|
||||||
const uint64_t readyTime = elapsed > measured ? elapsed - measured : 0;
|
const uint64_t readyTime = elapsed > measured ? elapsed - measured : 0;
|
||||||
|
|
||||||
sc->m_postProcessors[candidateIndex].RecordTiming(
|
// Use matching wall-clock boundaries for both modes. The split excludes the
|
||||||
fbRes->GetTimingEffectIndex(), fbRes->GetTimingToken(),
|
// cadence hold while including the indirect CPU copy only when it occurs.
|
||||||
fbRes->IsFullCopy(), postProcessTime + copyTime);
|
const uint64_t timingToken = fbRes->GetTimingToken();
|
||||||
|
if (timingToken && timingStart && prepareReady >= timingStart &&
|
||||||
|
readyEnd >= publishStart)
|
||||||
|
{
|
||||||
|
const uint64_t totalTime =
|
||||||
|
(prepareReady - timingStart) + (readyEnd - publishStart);
|
||||||
|
sc->m_postProcessors[candidateIndex].RecordTiming(
|
||||||
|
fbRes->GetTimingEffectIndex(), timingToken,
|
||||||
|
fbRes->IsFullCopy(), totalTime);
|
||||||
|
}
|
||||||
sc->m_devContext->RecordFrameTiming(readyEnd - publishStart);
|
sc->m_devContext->RecordFrameTiming(readyEnd - publishStart);
|
||||||
sc->m_devContext->SetFrameTiming(fbRes->GetFrameIndex(),
|
sc->m_devContext->SetFrameTiming(fbRes->GetFrameIndex(),
|
||||||
fbRes->GetCaptureTime(), postProcessTime, copyTime, readyTime);
|
fbRes->GetCaptureTime(), postProcessTime, copyTime, readyTime);
|
||||||
@@ -943,9 +954,10 @@ void CSwapChainProcessor::ReleaseCandidate(unsigned candidateIndex)
|
|||||||
static bool ResourceDescMatches(
|
static bool ResourceDescMatches(
|
||||||
const D3D12_RESOURCE_DESC& left, const D3D12_RESOURCE_DESC& right)
|
const D3D12_RESOURCE_DESC& left, const D3D12_RESOURCE_DESC& right)
|
||||||
{
|
{
|
||||||
|
// Alignment is allocation metadata. GetDesc may report the resolved value
|
||||||
|
// when the creation descriptor requested automatic alignment.
|
||||||
return
|
return
|
||||||
left.Dimension == right.Dimension &&
|
left.Dimension == right.Dimension &&
|
||||||
left.Alignment == right.Alignment &&
|
|
||||||
left.Width == right.Width &&
|
left.Width == right.Width &&
|
||||||
left.Height == right.Height &&
|
left.Height == right.Height &&
|
||||||
left.DepthOrArraySize == right.DepthOrArraySize &&
|
left.DepthOrArraySize == right.DepthOrArraySize &&
|
||||||
@@ -963,7 +975,7 @@ bool CSwapChainProcessor::EnsureCandidateResource(
|
|||||||
FrameCandidate& candidate = m_candidates[candidateIndex];
|
FrameCandidate& candidate = m_candidates[candidateIndex];
|
||||||
D3D12_RESOURCE_DESC desc = source->GetDesc();
|
D3D12_RESOURCE_DESC desc = source->GetDesc();
|
||||||
desc.Alignment = 0;
|
desc.Alignment = 0;
|
||||||
desc.Flags = static_cast<D3D12_RESOURCE_FLAGS>(
|
desc.Flags = static_cast<D3D12_RESOURCE_FLAGS>(
|
||||||
static_cast<UINT>(desc.Flags) &
|
static_cast<UINT>(desc.Flags) &
|
||||||
~static_cast<UINT>(D3D12_RESOURCE_FLAG_ALLOW_CROSS_ADAPTER));
|
~static_cast<UINT>(D3D12_RESOURCE_FLAG_ALLOW_CROSS_ADAPTER));
|
||||||
|
|
||||||
@@ -1599,6 +1611,9 @@ bool CSwapChainProcessor::SwapChainNewFrame(ComPtr<IDXGIResource> acquiredBuffer
|
|||||||
SetFullPendingDamage();
|
SetFullPendingDamage();
|
||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
|
// Candidate and copy-slot acquisition are common to both benchmark modes.
|
||||||
|
const uint64_t timingStart = timingToken ?
|
||||||
|
CFrameScheduler::Nanotime() : 0;
|
||||||
|
|
||||||
ComPtr<ID3D12Resource> copySrcResource = srcRes->GetRes();
|
ComPtr<ID3D12Resource> copySrcResource = srcRes->GetRes();
|
||||||
CD3D12CommandSlot * computeSlot = nullptr;
|
CD3D12CommandSlot * computeSlot = nullptr;
|
||||||
@@ -1677,23 +1692,24 @@ bool CSwapChainProcessor::SwapChainNewFrame(ComPtr<IDXGIResource> acquiredBuffer
|
|||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
|
|
||||||
candidate.srcFormat = srcFormat;
|
candidate.srcFormat = srcFormat;
|
||||||
candidate.dstFormat = dstFormat;
|
candidate.dstFormat = dstFormat;
|
||||||
candidate.nbDirtyRects = nbDirtyRects;
|
candidate.nbDirtyRects = nbDirtyRects;
|
||||||
candidate.pitch = postProcessor.GetOutputPitch();
|
candidate.pitch = postProcessor.GetOutputPitch();
|
||||||
candidate.frameSize = postProcessor.GetOutputSize();
|
candidate.frameSize = postProcessor.GetOutputSize();
|
||||||
candidate.captureTime = captureTime;
|
candidate.captureTime = captureTime;
|
||||||
candidate.postProcessStart = postProcessStart;
|
candidate.postProcessStart = postProcessStart;
|
||||||
candidate.prepareCopyStart = CFrameScheduler::Nanotime();
|
candidate.prepareCopyStart = CFrameScheduler::Nanotime();
|
||||||
candidate.prepareReady = 0;
|
candidate.prepareReady = 0;
|
||||||
candidate.prepareGPUStart = 0;
|
candidate.prepareGPUStart = 0;
|
||||||
candidate.prepareGPUEnd = 0;
|
candidate.prepareGPUEnd = 0;
|
||||||
|
candidate.timingStart = timingStart;
|
||||||
candidate.prepareTimingValid = false;
|
candidate.prepareTimingValid = false;
|
||||||
if (nbDirtyRects)
|
if (nbDirtyRects)
|
||||||
memcpy(candidate.dirtyRects, currentDirtyRects,
|
memcpy(candidate.dirtyRects, currentDirtyRects,
|
||||||
nbDirtyRects * sizeof(*candidate.dirtyRects));
|
nbDirtyRects * sizeof(*candidate.dirtyRects));
|
||||||
candidate.timingEffectIndex = timingEffectIndex;
|
candidate.timingEffectIndex = timingEffectIndex;
|
||||||
candidate.timingToken = timingToken;
|
candidate.timingToken = timingToken;
|
||||||
|
|
||||||
copySlot->SetCompletionCallback(
|
copySlot->SetCompletionCallback(
|
||||||
&CandidateCompletionFunction, this, &candidate);
|
&CandidateCompletionFunction, this, &candidate);
|
||||||
|
|||||||
@@ -66,24 +66,25 @@ private:
|
|||||||
|
|
||||||
struct FrameCandidate
|
struct FrameCandidate
|
||||||
{
|
{
|
||||||
CandidateState state = CANDIDATE_FREE;
|
CandidateState state = CANDIDATE_FREE;
|
||||||
ComPtr<ID3D12Resource> resource;
|
ComPtr<ID3D12Resource> resource;
|
||||||
D12FrameFormat srcFormat = {};
|
D12FrameFormat srcFormat = {};
|
||||||
D12FrameFormat dstFormat = {};
|
D12FrameFormat dstFormat = {};
|
||||||
RECT dirtyRects[LG_MAX_DIRTY_RECTS] = {};
|
RECT dirtyRects[LG_MAX_DIRTY_RECTS] = {};
|
||||||
unsigned nbDirtyRects = 0;
|
unsigned nbDirtyRects = 0;
|
||||||
unsigned pitch = 0;
|
unsigned pitch = 0;
|
||||||
size_t frameSize = 0;
|
size_t frameSize = 0;
|
||||||
uint64_t sequence = 0;
|
uint64_t sequence = 0;
|
||||||
uint64_t captureTime = 0;
|
uint64_t captureTime = 0;
|
||||||
uint64_t postProcessStart = 0;
|
uint64_t postProcessStart = 0;
|
||||||
uint64_t prepareCopyStart = 0;
|
uint64_t prepareCopyStart = 0;
|
||||||
uint64_t prepareReady = 0;
|
uint64_t prepareReady = 0;
|
||||||
uint64_t prepareGPUStart = 0;
|
uint64_t prepareGPUStart = 0;
|
||||||
uint64_t prepareGPUEnd = 0;
|
uint64_t prepareGPUEnd = 0;
|
||||||
unsigned timingEffectIndex = 0;
|
uint64_t timingStart = 0;
|
||||||
uint64_t timingToken = 0;
|
unsigned timingEffectIndex = 0;
|
||||||
bool prepareTimingValid = false;
|
uint64_t timingToken = 0;
|
||||||
|
bool prepareTimingValid = false;
|
||||||
};
|
};
|
||||||
|
|
||||||
struct CandidateDamageTail
|
struct CandidateDamageTail
|
||||||
|
|||||||
Reference in New Issue
Block a user