[idd] lgmp: stream staged copies to WC memory

This commit is contained in:
Geoffrey McRae
2026-08-17 01:27:38 +10:00
parent 636e55db0d
commit 26cafbe3df
6 changed files with 284 additions and 7 deletions

View File

@@ -74,6 +74,10 @@
<ClCompile Include="WCCopyAVX2.cpp">
<EnableEnhancedInstructionSet>AdvancedVectorExtensions2</EnableEnhancedInstructionSet>
</ClCompile>
<ClCompile Include="WCWrite.cpp" />
<ClCompile Include="WCWriteAVX2.cpp">
<EnableEnhancedInstructionSet>AdvancedVectorExtensions2</EnableEnhancedInstructionSet>
</ClCompile>
</ItemGroup>
<ItemGroup>
<ClInclude Include="CClipboardChannel.h" />
@@ -90,6 +94,7 @@
<ClInclude Include="RefreshRate.h" />
<ClInclude Include="Seq.h" />
<ClInclude Include="WCCopy.h" />
<ClInclude Include="WCWrite.h" />
</ItemGroup>
<Import Project="$(VCTargetsPath)\Microsoft.Cpp.targets" />
<ImportGroup Label="ExtensionTargets" />

View File

@@ -33,6 +33,12 @@
<ClCompile Include="WCCopyAVX2.cpp">
<Filter>Source Files</Filter>
</ClCompile>
<ClCompile Include="WCWrite.cpp">
<Filter>Source Files</Filter>
</ClCompile>
<ClCompile Include="WCWriteAVX2.cpp">
<Filter>Source Files</Filter>
</ClCompile>
</ItemGroup>
<ItemGroup>
<ClInclude Include="Atomic.h">
@@ -77,5 +83,8 @@
<ClInclude Include="WCCopy.h">
<Filter>Header Files</Filter>
</ClInclude>
<ClInclude Include="WCWrite.h">
<Filter>Header Files</Filter>
</ClInclude>
</ItemGroup>
</Project>

146
idd/LGCommon/WCWrite.cpp Normal file
View File

@@ -0,0 +1,146 @@
/**
* Looking Glass
* Copyright © 2017-2026 The Looking Glass Authors
* https://looking-glass.io
*
* This program is free software; you can redistribute it and/or modify it
* under the terms of the GNU General Public License as published by the Free
* Software Foundation; either version 2 of the License, or (at your option)
* any later version.
*
* This program is distributed in the hope that it will be useful, but WITHOUT
* ANY WARRANTY; without even the implied warranty of MERCHANTABILITY or
* FITNESS FOR A PARTICULAR PURPOSE. See the GNU General Public License for
* more details.
*
* You should have received a copy of the GNU General Public License along
* with this program; if not, write to the Free Software Foundation, Inc., 59
* Temple Place, Suite 330, Boston, MA 02111-1307 USA
*/
#include "WCWrite.h"
#include <atomic>
#include <immintrin.h>
#include <intrin.h>
#include <stdint.h>
#include <string.h>
namespace WCWrite
{
namespace Detail
{
void CopyAVX2(void * destination, const void * source, size_t size);
}
namespace
{
using CopyFn = void (*)(void *, const void *, size_t);
static constexpr unsigned __int64 AVX_XSTATE_MASK = 0x6U;
void CopyFallback(void * destination, const void * source, size_t size)
{
memcpy(destination, source, size);
}
#if defined(_M_IX86) || defined(_M_X64)
__declspec(noinline) void CopySSE41(
void * destination, const void * source, size_t size)
{
uint8_t * dst = static_cast<uint8_t *>(destination);
const uint8_t * src = static_cast<const uint8_t *>(source);
const size_t prefix =
(16U - (reinterpret_cast<uintptr_t>(dst) & 15U)) & 15U;
if (prefix)
{
const size_t copy = prefix < size ? prefix : size;
memcpy(dst, src, copy);
dst += copy;
src += copy;
size -= copy;
}
while (size >= 64U)
{
const __m128i v0 = _mm_loadu_si128(
reinterpret_cast<const __m128i *>(src + 0));
const __m128i v1 = _mm_loadu_si128(
reinterpret_cast<const __m128i *>(src + 16));
const __m128i v2 = _mm_loadu_si128(
reinterpret_cast<const __m128i *>(src + 32));
const __m128i v3 = _mm_loadu_si128(
reinterpret_cast<const __m128i *>(src + 48));
__m128i * output = reinterpret_cast<__m128i *>(dst);
_mm_stream_si128(output + 0, v0);
_mm_stream_si128(output + 1, v1);
_mm_stream_si128(output + 2, v2);
_mm_stream_si128(output + 3, v3);
dst += 64U;
src += 64U;
size -= 64U;
}
while (size >= 16U)
{
const __m128i value = _mm_loadu_si128(
reinterpret_cast<const __m128i *>(src));
_mm_stream_si128(reinterpret_cast<__m128i *>(dst), value);
dst += 16U;
src += 16U;
size -= 16U;
}
if (size)
memcpy(dst, src, size);
}
CopyFn SelectCopy()
{
int registers[4] = {};
__cpuid(registers, 0);
const int maximumLeaf = registers[0];
if (maximumLeaf < 1)
return &CopyFallback;
__cpuidex(registers, 1, 0);
const bool sse41 = (registers[2] & (1 << 19)) != 0;
const bool avx = (registers[2] & (1 << 28)) != 0;
const bool osxsave = (registers[2] & (1 << 27)) != 0;
if (maximumLeaf >= 7 && avx && osxsave &&
(_xgetbv(0) & AVX_XSTATE_MASK) == AVX_XSTATE_MASK)
{
__cpuidex(registers, 7, 0);
if ((registers[1] & (1 << 5)) != 0)
return &Detail::CopyAVX2;
}
return sse41 ? &CopySSE41 : &CopyFallback;
}
#else
CopyFn SelectCopy()
{
return &CopyFallback;
}
#endif
}
void Copy(void * destination, const void * source, size_t size)
{
if (!size)
return;
static const CopyFn copy = SelectCopy();
copy(destination, source, size);
}
void Flush()
{
#if defined(_M_IX86) || defined(_M_X64)
_mm_sfence();
#else
std::atomic_thread_fence(std::memory_order_release);
#endif
}
}

32
idd/LGCommon/WCWrite.h Normal file
View File

@@ -0,0 +1,32 @@
/**
* Looking Glass
* Copyright © 2017-2026 The Looking Glass Authors
* https://looking-glass.io
*
* This program is free software; you can redistribute it and/or modify it
* under the terms of the GNU General Public License as published by the Free
* Software Foundation; either version 2 of the License, or (at your option)
* any later version.
*
* This program is distributed in the hope that it will be useful, but WITHOUT
* ANY WARRANTY; without even the implied warranty of MERCHANTABILITY or
* FITNESS FOR A PARTICULAR PURPOSE. See the GNU General Public License for
* more details.
*
* You should have received a copy of the GNU General Public License along
* with this program; if not, write to the Free Software Foundation, Inc., 59
* Temple Place, Suite 330, Boston, MA 02111-1307 USA
*/
#pragma once
#include <stddef.h>
namespace WCWrite
{
// Copy from normal cacheable memory into write-combined memory. The source
// and destination ranges must not overlap. Call Flush before publishing the
// written range to another processor or device.
void Copy(void * destination, const void * source, size_t size);
void Flush();
}

View File

@@ -0,0 +1,84 @@
/**
* Looking Glass
* Copyright © 2017-2026 The Looking Glass Authors
* https://looking-glass.io
*
* This program is free software; you can redistribute it and/or modify it
* under the terms of the GNU General Public License as published by the Free
* Software Foundation; either version 2 of the License, or (at your option)
* any later version.
*
* This program is distributed in the hope that it will be useful, but WITHOUT
* ANY WARRANTY; without even the implied warranty of MERCHANTABILITY or
* FITNESS FOR A PARTICULAR PURPOSE. See the GNU General Public License for
* more details.
*
* You should have received a copy of the GNU General Public License along
* with this program; if not, write to the Free Software Foundation, Inc., 59
* Temple Place, Suite 330, Boston, MA 02111-1307 USA
*/
#include "WCWrite.h"
#include <immintrin.h>
#include <stdint.h>
#include <string.h>
namespace WCWrite
{
namespace Detail
{
__declspec(noinline) void CopyAVX2(
void * destination, const void * source, size_t size)
{
uint8_t * dst = static_cast<uint8_t *>(destination);
const uint8_t * src = static_cast<const uint8_t *>(source);
const size_t prefix =
(32U - (reinterpret_cast<uintptr_t>(dst) & 31U)) & 31U;
if (prefix)
{
const size_t copy = prefix < size ? prefix : size;
memcpy(dst, src, copy);
dst += copy;
src += copy;
size -= copy;
}
while (size >= 128U)
{
const __m256i v0 = _mm256_loadu_si256(
reinterpret_cast<const __m256i *>(src + 0));
const __m256i v1 = _mm256_loadu_si256(
reinterpret_cast<const __m256i *>(src + 32));
const __m256i v2 = _mm256_loadu_si256(
reinterpret_cast<const __m256i *>(src + 64));
const __m256i v3 = _mm256_loadu_si256(
reinterpret_cast<const __m256i *>(src + 96));
__m256i * output = reinterpret_cast<__m256i *>(dst);
_mm256_stream_si256(output + 0, v0);
_mm256_stream_si256(output + 1, v1);
_mm256_stream_si256(output + 2, v2);
_mm256_stream_si256(output + 3, v3);
dst += 128U;
src += 128U;
size -= 128U;
}
while (size >= 32U)
{
const __m256i value = _mm256_loadu_si256(
reinterpret_cast<const __m256i *>(src));
_mm256_stream_si256(reinterpret_cast<__m256i *>(dst), value);
dst += 32U;
src += 32U;
size -= 32U;
}
if (size)
memcpy(dst, src, size);
_mm256_zeroupper();
}
}
}

View File

@@ -25,6 +25,7 @@
#include "transport/lgmp/CLGMPHost.h"
#include "Atomic.h"
#include "CDebug.h"
#include "WCWrite.h"
#include <cstring>
#include <memory>
@@ -1335,15 +1336,14 @@ void CLGMPFrameTransport::WriteFrameBuffer(unsigned frameIndex, void * src,
{
LGMPBuffer * fb = m_frameBuffer[frameIndex];
memcpy(
reinterpret_cast<void *>(
reinterpret_cast<uintptr_t>(fb->data) + offset),
reinterpret_cast<void *>(
reinterpret_cast<uintptr_t>(src) + offset),
len);
WCWrite::Copy(fb->data + offset,
static_cast<uint8_t *>(src) + offset, len);
if (setWritePos)
{
WCWrite::Flush();
fb->wp = (uint32_t)(offset + len);
}
}
void CLGMPFrameTransport::WriteFrameBufferRows(unsigned frameIndex,
@@ -1355,7 +1355,7 @@ void CLGMPFrameTransport::WriteFrameBufferRows(unsigned frameIndex,
uint8_t * source = static_cast<uint8_t *>(src) + offset;
for (unsigned row = 0; row < rows; ++row)
{
memcpy(dst, source, rowBytes);
WCWrite::Copy(dst, source, rowBytes);
dst += pitch;
source += pitch;
}
@@ -1365,5 +1365,6 @@ void CLGMPFrameTransport::FinalizeFrameBuffer(unsigned frameIndex) const
{
const KVMFRFrame * frame = m_frame[frameIndex];
LGMPBuffer * fb = m_frameBuffer[frameIndex];
WCWrite::Flush();
fb->wp = frame->dataHeight * frame->pitch;
}