[idd] scheduler: retain frames behind stalled owners

Keep preparing the newest frame when both private owner lanes remain
referenced by a non-consuming timing client. Reuse only an unreferenced
completed buffer and continue serving eligible shared subscribers.

Track owner delivery separately from local retention so cadence advances
without acknowledging owner delivery or accepting phase feedback. Republish
the newest completed frame as soon as an owner lane becomes available.
This commit is contained in:
Geoffrey McRae
2026-08-06 16:56:25 +10:00
parent 05fe4f1dd8
commit ece6afd74e
5 changed files with 210 additions and 48 deletions

View File

@@ -726,6 +726,44 @@ void CFrameScheduler::FramePublished(const Schedule& schedule,
ReleaseSRWLockExclusive(&m_lock);
}
void CFrameScheduler::FrameRetained(const Schedule& schedule,
uint64_t now, bool periodic)
{
bool wake = false;
AcquireSRWLockExclusive(&m_lock);
if (m_scheduling && schedule.clientID == m_schedule.clientID &&
schedule.generation == m_schedule.generation &&
schedule.epoch == m_schedule.epoch &&
schedule.deadlineSerial == m_deadlineSerial &&
schedule.deadline == m_nextDeadline)
{
if (schedule.forceTicket > m_forceAckTicket)
m_forceAckTicket = schedule.forceTicket;
// The submission is retained locally, but the owner has not seen it.
// Keep one republish request pending until an owner lane is released.
if (m_republishRequestTicket == m_republishAckTicket)
{
++m_republishRequestTicket;
wake = true;
}
if (periodic)
{
AdvanceCurrentDeadline();
AdvanceDeadline(now);
}
Client * client = FindClient(m_schedule.clientID);
if (client)
client->nextDelivery = m_nextDeadline;
}
ReleaseSRWLockExclusive(&m_lock);
if (wake)
WakePublisher();
}
bool CFrameScheduler::TryFrameCompleted(const Schedule& schedule,
uint32_t frameSerial, uint64_t completedAt)
{

View File

@@ -146,6 +146,8 @@ public:
bool TryFrameSubmitted(const Schedule& schedule, uint32_t frameSerial);
void FramePublished(const Schedule& schedule, uint32_t frameSerial,
uint64_t now, bool periodic);
void FrameRetained(const Schedule& schedule, uint64_t now,
bool periodic);
void FrameRepublished(const Schedule& schedule, uint32_t frameSerial);
bool TryFrameCompleted(const Schedule& schedule, uint32_t frameSerial,
uint64_t completedAt);

View File

@@ -1185,11 +1185,13 @@ bool CIndirectDeviceContext::SetupLGMP(size_t alignSize)
m_frame[i]->offset = (uint32_t)alignOffset;
m_frameBuffer[i] = (FrameBuffer*)(((uint8_t*)m_frame[i]) + alignOffset);
m_frameInFlight[i].store(false, std::memory_order_release);
m_frameCompleted[i] = false;
}
m_maxFrameSize = maxFrameSize;
m_submittedFrameIndex.store(-1, std::memory_order_release);
m_readyFrameIndex.store(-1, std::memory_order_release);
m_deferredOwnerFrameIndex = -1;
m_framePublishSequence = 0;
memset(m_frameLastPublishSequence, 0,
sizeof(m_frameLastPublishSequence));
@@ -1245,9 +1247,11 @@ void CIndirectDeviceContext::DeInitLGMP()
m_frameScheduler.Reset();
m_submittedFrameIndex.store(-1, std::memory_order_release);
m_readyFrameIndex.store(-1, std::memory_order_release);
m_deferredOwnerFrameIndex = -1;
m_framePublishSequence = 0;
memset(m_frameLastPublishSequence, 0,
sizeof(m_frameLastPublishSequence));
memset(m_frameCompleted, 0, sizeof(m_frameCompleted));
for (FrameDelivery& delivery : m_frameDelivery)
delivery = {};
for (OwnerDelivery& delivery : m_ownerDelivery)
@@ -1266,9 +1270,11 @@ void CIndirectDeviceContext::DeInitLGMP()
AcquireSRWLockExclusive(&m_framePublishLock);
m_submittedFrameIndex.store(-1, std::memory_order_release);
m_readyFrameIndex.store(-1, std::memory_order_release);
m_deferredOwnerFrameIndex = -1;
m_framePublishSequence = 0;
memset(m_frameLastPublishSequence, 0,
sizeof(m_frameLastPublishSequence));
memset(m_frameCompleted, 0, sizeof(m_frameCompleted));
for (FrameDelivery& delivery : m_frameDelivery)
delivery = {};
for (OwnerDelivery& delivery : m_ownerDelivery)
@@ -1646,7 +1652,31 @@ bool CIndirectDeviceContext::HasMatchingOwnerDelivery(
return false;
}
int CIndirectDeviceContext::FindAvailableFrameBuffer() const
bool CIndirectDeviceContext::FrameBufferReferenced(
unsigned frameIndex) const
{
if (frameIndex >= LGMP_Q_FRAME_BUFFER_LEN)
return true;
const FrameDelivery& delivery = m_frameDelivery[frameIndex];
if (delivery.ownerQueueMask || delivery.sharedPending ||
delivery.sharedOwnerPending ||
lgmpHostQueuePayloadPending(
m_frameQueue, m_frameMemory[frameIndex]))
return true;
for (unsigned queueIndex = 0;
queueIndex < LGMP_Q_FRAME_LEN;
++queueIndex)
if (lgmpHostQueuePayloadPending(
m_frameOwnerQueue[queueIndex], m_frameMemory[frameIndex]))
return true;
return false;
}
int CIndirectDeviceContext::FindAvailableFrameBuffer(
bool allowReady) const
{
const LONG readyFrameIndex =
m_readyFrameIndex.load(std::memory_order_acquire);
@@ -1658,20 +1688,7 @@ int CIndirectDeviceContext::FindAvailableFrameBuffer() const
{
if (static_cast<LONG>(frameIndex) == readyFrameIndex ||
m_frameInFlight[frameIndex].load(std::memory_order_acquire) ||
m_frameDelivery[frameIndex].ownerQueueMask ||
m_frameDelivery[frameIndex].sharedPending ||
lgmpHostQueuePayloadPending(
m_frameQueue, m_frameMemory[frameIndex]))
continue;
bool ownerPending = false;
for (unsigned queueIndex = 0;
!ownerPending && queueIndex < LGMP_Q_FRAME_LEN;
++queueIndex)
ownerPending = lgmpHostQueuePayloadPending(
m_frameOwnerQueue[queueIndex], m_frameMemory[frameIndex]);
if (ownerPending)
FrameBufferReferenced(frameIndex))
continue;
if (available < 0 ||
@@ -1682,9 +1699,42 @@ int CIndirectDeviceContext::FindAvailableFrameBuffer() const
}
}
if (available >= 0 || !allowReady || readyFrameIndex < 0 ||
m_frameInFlight[readyFrameIndex].load(std::memory_order_acquire) ||
FrameBufferReferenced(static_cast<unsigned>(readyFrameIndex)))
return available;
// A blocked owner can consume the other two buffers indefinitely. Once
// the retained frame has no queue references it is safe to replace it with
// a newer frame rather than waiting for the owner's LGMP timeout.
available = static_cast<int>(readyFrameIndex);
return available;
}
int CIndirectDeviceContext::FindNewestCompletedFrame(
unsigned excludeFrameIndex) const
{
int newestFrame = -1;
uint64_t newestSequence = 0;
for (unsigned frameIndex = 0;
frameIndex < LGMP_Q_FRAME_BUFFER_LEN;
++frameIndex)
{
if (frameIndex == excludeFrameIndex || !m_frameCompleted[frameIndex] ||
m_frameInFlight[frameIndex].load(std::memory_order_acquire))
continue;
if (newestFrame < 0 ||
m_frameLastPublishSequence[frameIndex] > newestSequence)
{
newestFrame = static_cast<int>(frameIndex);
newestSequence = m_frameLastPublishSequence[frameIndex];
}
}
return newestFrame;
}
bool CIndirectDeviceContext::FrameBufferAvailable(
const CFrameScheduler::Schedule& schedule)
{
@@ -1696,19 +1746,24 @@ bool CIndirectDeviceContext::FrameBufferAvailable(
AcquireSRWLockShared(&m_framePublishLock);
bool deliveryAvailable;
bool ownerBlocked = false;
// Pipeline one frame through each independent owner lane. Count the shared
// fallback against the same limit so it cannot become a third delivery for
// the same owner.
// the same owner. Once both are occupied, a fully unreferenced buffer can
// still retain a newer frame for secondary delivery and later republish.
if (schedule.clientID)
deliveryAvailable =
CountOwnerDeliveries(schedule.clientID) < LGMP_Q_FRAME_LEN &&
(FindAvailableOwnerQueue(0) >= 0 ||
lgmpHostQueuePending(m_frameQueue) < LGMP_Q_FRAME_LEN);
{
ownerBlocked =
CountOwnerDeliveries(schedule.clientID) >= LGMP_Q_FRAME_LEN;
deliveryAvailable = ownerBlocked ||
FindAvailableOwnerQueue(0) >= 0 ||
lgmpHostQueuePending(m_frameQueue) < LGMP_Q_FRAME_LEN;
}
else
deliveryAvailable = lgmpHostQueuePending(m_frameQueue) == 0;
const bool available = deliveryAvailable &&
FindAvailableFrameBuffer() >= 0;
FindAvailableFrameBuffer(ownerBlocked) >= 0;
ReleaseSRWLockShared(&m_framePublishLock);
return available;
}
@@ -1784,8 +1839,9 @@ bool CIndirectDeviceContext::ReplaySharedFrame(uint64_t now, bool& retry)
}
CIndirectDeviceContext::PreparedFrameBuffer CIndirectDeviceContext::PrepareFrameBuffer(
unsigned pitch, const D12FrameFormat& srcFormat, const D12FrameFormat& dstFormat,
const RECT * dirtyRects, unsigned nbDirtyRects)
unsigned pitch, const D12FrameFormat& srcFormat,
const D12FrameFormat& dstFormat, const RECT * dirtyRects,
unsigned nbDirtyRects, const CFrameScheduler::Schedule& schedule)
{
PreparedFrameBuffer result = {};
@@ -1801,13 +1857,26 @@ CIndirectDeviceContext::PreparedFrameBuffer CIndirectDeviceContext::PrepareFrame
}
AcquireSRWLockExclusive(&m_framePublishLock);
const int availableFrameIndex = FindAvailableFrameBuffer();
const bool ownerBlocked = schedule.clientID &&
CountOwnerDeliveries(schedule.clientID) >= LGMP_Q_FRAME_LEN;
const int availableFrameIndex =
FindAvailableFrameBuffer(ownerBlocked);
bool expected = false;
const bool acquired = availableFrameIndex >= 0 &&
m_frameInFlight[availableFrameIndex].compare_exchange_strong(
expected, true, std::memory_order_acq_rel);
if (acquired)
{
const LONG readyFrameIndex =
m_readyFrameIndex.load(std::memory_order_acquire);
if (availableFrameIndex == readyFrameIndex)
m_readyFrameIndex.store(
FindNewestCompletedFrame(
static_cast<unsigned>(availableFrameIndex)),
std::memory_order_release);
m_frameCompleted[availableFrameIndex] = false;
m_frameDelivery[availableFrameIndex] = {};
}
const bool fullCopy = acquired &&
(!m_frameLastPublishSequence[availableFrameIndex] ||
m_framePublishSequence >
@@ -1952,8 +2021,9 @@ CIndirectDeviceContext::PreparedFrameBuffer CIndirectDeviceContext::PrepareFrame
}
bool CIndirectDeviceContext::PublishFrameBuffer(unsigned frameIndex,
const CFrameScheduler::Schedule& schedule)
const CFrameScheduler::Schedule& schedule, bool& deliveredToOwner)
{
deliveredToOwner = false;
if (!m_frameQueue || frameIndex >= LGMP_Q_FRAME_BUFFER_LEN)
return false;
@@ -1980,12 +2050,21 @@ bool CIndirectDeviceContext::PublishFrameBuffer(unsigned frameIndex,
if (schedule.clientID)
{
if (CountOwnerDeliveries(schedule.clientID) >= LGMP_Q_FRAME_LEN)
status = LGMP_ERR_QUEUE_FULL;
{
// Both owner lanes still reference older frames. Retain the newest
// frame locally and serve any unblocked secondary clients without
// violating the lifetime of those outstanding LGMP payloads.
PostSharedFrame(frameIndex, schedule.clientID, now);
published = true;
}
else
{
const int ownerQueueIndex = FindAvailableOwnerQueue(frameIndex);
if (ownerQueueIndex < 0)
{
published = PostSharedOwnerFrame(frameIndex, schedule);
deliveredToOwner = published;
}
else
{
unsigned recipientCount = 0;
@@ -2005,7 +2084,8 @@ bool CIndirectDeviceContext::PublishFrameBuffer(unsigned frameIndex,
FrameDelivery& delivery = m_frameDelivery[frameIndex];
delivery.ownerQueueMask |= 1U << queueIndex;
published = true;
deliveredToOwner = true;
published = true;
PostSharedFrame(frameIndex, schedule.clientID, now);
}
@@ -2013,11 +2093,17 @@ bool CIndirectDeviceContext::PublishFrameBuffer(unsigned frameIndex,
}
}
else
published = PostSharedFrame(frameIndex, 0, now) != SHARED_FRAME_FAILED;
{
published = PostSharedFrame(
frameIndex, 0, now) != SHARED_FRAME_FAILED;
deliveredToOwner = published;
}
if (published)
{
m_frameLastPublishSequence[frameIndex] = ++m_framePublishSequence;
m_deferredOwnerFrameIndex = schedule.clientID && !deliveredToOwner ?
static_cast<LONG>(frameIndex) : -1;
m_submittedFrameIndex.store(
static_cast<LONG>(frameIndex), std::memory_order_release);
}
@@ -2049,8 +2135,16 @@ bool CIndirectDeviceContext::RepublishFrameBuffer(
return false;
}
const LONG frameIndex =
m_readyFrameIndex.load(std::memory_order_acquire);
LONG frameIndex = m_deferredOwnerFrameIndex;
if (frameIndex >= 0 &&
!m_frameCompleted[frameIndex] &&
!m_frameInFlight[frameIndex].load(std::memory_order_acquire))
{
m_deferredOwnerFrameIndex = -1;
frameIndex = -1;
}
if (frameIndex < 0)
frameIndex = m_readyFrameIndex.load(std::memory_order_acquire);
if (frameIndex < 0 ||
m_frameInFlight[frameIndex].load(std::memory_order_acquire))
{
@@ -2066,6 +2160,8 @@ bool CIndirectDeviceContext::RepublishFrameBuffer(
if (HasMatchingOwnerDelivery(schedule.clientID,
static_cast<unsigned>(frameIndex), scheduleToken))
{
if (m_deferredOwnerFrameIndex == frameIndex)
m_deferredOwnerFrameIndex = -1;
ReleaseSRWLockExclusive(&m_framePublishLock);
m_frameScheduler.FrameRepublished(schedule, frameSerial);
return true;
@@ -2083,6 +2179,8 @@ bool CIndirectDeviceContext::RepublishFrameBuffer(
{
const bool published = PostSharedOwnerFrame(
static_cast<unsigned>(frameIndex), deliverySchedule);
if (published && m_deferredOwnerFrameIndex == frameIndex)
m_deferredOwnerFrameIndex = -1;
ReleaseSRWLockExclusive(&m_framePublishLock);
if (published)
m_frameScheduler.FrameRepublished(schedule, frameSerial);
@@ -2105,6 +2203,8 @@ bool CIndirectDeviceContext::RepublishFrameBuffer(
FrameDelivery& delivery = m_frameDelivery[frameIndex];
delivery.ownerQueueMask |= 1U << queueIndex;
if (m_deferredOwnerFrameIndex == frameIndex)
m_deferredOwnerFrameIndex = -1;
}
ReleaseSRWLockExclusive(&m_framePublishLock);
@@ -2121,14 +2221,18 @@ bool CIndirectDeviceContext::RepublishFrameBuffer(
}
void CIndirectDeviceContext::CommitFrameBuffer(unsigned frameIndex,
const CFrameScheduler::Schedule& schedule, bool periodic)
const CFrameScheduler::Schedule& schedule, bool periodic,
bool deliveredToOwner)
{
if (frameIndex >= LGMP_Q_FRAME_BUFFER_LEN)
return;
m_frameScheduler.FramePublished(
schedule, m_frame[frameIndex]->frameSerial,
CFrameScheduler::Nanotime(), periodic);
const uint64_t now = CFrameScheduler::Nanotime();
if (deliveredToOwner)
m_frameScheduler.FramePublished(
schedule, m_frame[frameIndex]->frameSerial, now, periodic);
else
m_frameScheduler.FrameRetained(schedule, now, periodic);
}
bool CIndirectDeviceContext::TryFrameSubmitted(unsigned frameIndex,
@@ -2184,6 +2288,9 @@ void CIndirectDeviceContext::AbortFrameBuffer(unsigned frameIndex)
m_frameBuffer[frameIndex]->wp = 0;
InterlockedExchange(
(volatile LONG *)&m_frame[frameIndex]->timingValid, 0);
m_frameCompleted[frameIndex] = false;
if (m_deferredOwnerFrameIndex == static_cast<LONG>(frameIndex))
m_deferredOwnerFrameIndex = -1;
m_frameInFlight[frameIndex].store(false, std::memory_order_release);
ReleaseSRWLockExclusive(&m_framePublishLock);
}
@@ -2205,6 +2312,9 @@ void CIndirectDeviceContext::CompleteFrameBuffer(
return;
AcquireSRWLockExclusive(&m_framePublishLock);
m_frameCompleted[frameIndex] = succeeded;
if (!succeeded && m_deferredOwnerFrameIndex == static_cast<LONG>(frameIndex))
m_deferredOwnerFrameIndex = -1;
if (succeeded)
{
// Completion callbacks may run out of order. Never replace a newer ready

View File

@@ -110,11 +110,13 @@ private:
size_t m_alignSize = 0;
size_t m_frameMemoryOffset = 0;
size_t m_maxFrameSize = 0;
// LGMP publication precedes copy completion, so only the ready index is safe
// to retain and replay.
std::atomic<LONG> m_submittedFrameIndex = -1;
std::atomic<LONG> m_readyFrameIndex = -1;
// LGMP publication precedes copy completion. Replay only completed frames;
// the deferred index tracks the newest frame still owed to the owner.
std::atomic<LONG> m_submittedFrameIndex = -1;
std::atomic<LONG> m_readyFrameIndex = -1;
LONG m_deferredOwnerFrameIndex = -1;
std::atomic<bool> m_frameInFlight[LGMP_Q_FRAME_BUFFER_LEN] = {};
bool m_frameCompleted[LGMP_Q_FRAME_BUFFER_LEN] = {};
SRWLOCK m_framePublishLock = SRWLOCK_INIT;
uint64_t m_framePublishSequence = 0;
uint64_t m_frameLastPublishSequence[LGMP_Q_FRAME_BUFFER_LEN] = {};
@@ -190,7 +192,9 @@ private:
void DeInitLGMP();
void LGMPTimer();
void ProcessFrameDeliveries();
int FindAvailableFrameBuffer() const;
bool FrameBufferReferenced(unsigned frameIndex) const;
int FindAvailableFrameBuffer(bool allowReady) const;
int FindNewestCompletedFrame(unsigned excludeFrameIndex) const;
int FindAvailableOwnerQueue(unsigned preferredIndex) const;
unsigned CountOwnerDeliveries(uint32_t clientID) const;
bool HasMatchingOwnerDelivery(uint32_t clientID, unsigned frameIndex,
@@ -278,14 +282,18 @@ public:
void ProcessFrameQueue();
bool GetSharedFrameTarget(uint64_t now, uint64_t& target);
bool ReplaySharedFrame(uint64_t now, bool& retry);
PreparedFrameBuffer PrepareFrameBuffer(unsigned pitch, const D12FrameFormat& srcFormat, const D12FrameFormat& dstFormat, const RECT * dirtyRects, unsigned nbDirtyRects);
bool PublishFrameBuffer(unsigned frameIndex,
PreparedFrameBuffer PrepareFrameBuffer(unsigned pitch,
const D12FrameFormat& srcFormat, const D12FrameFormat& dstFormat,
const RECT * dirtyRects, unsigned nbDirtyRects,
const CFrameScheduler::Schedule& schedule);
bool PublishFrameBuffer(unsigned frameIndex,
const CFrameScheduler::Schedule& schedule, bool& deliveredToOwner);
bool RepublishFrameBuffer(const CFrameScheduler::Schedule& schedule);
bool TryFrameSubmitted(unsigned frameIndex,
const CFrameScheduler::Schedule& schedule);
void CommitFrameBuffer(unsigned frameIndex,
const CFrameScheduler::Schedule& schedule, bool periodic);
const CFrameScheduler::Schedule& schedule, bool periodic,
bool deliveredToOwner);
void AbortFrameBuffer(unsigned frameIndex);
void FailFrameBuffer(unsigned frameIndex);
void CompleteFrameBuffer(unsigned frameIndex, bool succeeded);

View File

@@ -1208,7 +1208,8 @@ bool CSwapChainProcessor::PublishNewestCandidate(
candidate.srcFormat,
candidate.dstFormat,
candidate.dirtyRects,
candidate.nbDirtyRects);
candidate.nbDirtyRects,
schedule);
if (!buffer.mem)
{
restoreCandidates();
@@ -1298,10 +1299,12 @@ bool CSwapChainProcessor::PublishNewestCandidate(
copyDirtyRects, nbCopyDirtyRects, fullCopy);
copySlot->EndTiming();
// Reserve the LGMP message before submitting the copy. This makes post
// failure recoverable without racing a very fast GPU completion callback.
// Reserve the LGMP delivery or retained-frame slot before submitting the
// copy. This makes failure recoverable without racing a very fast GPU
// completion callback.
bool deliveredToOwner;
if (!m_devContext->PublishFrameBuffer(
buffer.frameIndex, schedule))
buffer.frameIndex, schedule, deliveredToOwner))
{
copySlot->Cancel();
m_devContext->AbortFrameBuffer(buffer.frameIndex);
@@ -1311,7 +1314,8 @@ bool CSwapChainProcessor::PublishNewestCandidate(
CFrameScheduler::Schedule frameSchedule = schedule;
// Phase accounting must never hold up D3D submission. Keep the immutable
// delivery identity and discard only this feedback sample on contention.
if (!m_devContext->TryFrameSubmitted(buffer.frameIndex, schedule))
if (!deliveredToOwner ||
!m_devContext->TryFrameSubmitted(buffer.frameIndex, schedule))
frameSchedule.phaseEligible = false;
fbRes->SetSchedule(frameSchedule);
@@ -1356,7 +1360,7 @@ bool CSwapChainProcessor::PublishNewestCandidate(
}
m_devContext->CommitFrameBuffer(
buffer.frameIndex, schedule, periodic);
buffer.frameIndex, schedule, periodic, deliveredToOwner);
unsigned superseded = 0;
AcquireSRWLockExclusive(&m_candidateLock);