diff --git a/neo/engine/doomdll.vcxproj.user b/neo/engine/doomdll.vcxproj.user
index 08538da5..e4f905db 100644
--- a/neo/engine/doomdll.vcxproj.user
+++ b/neo/engine/doomdll.vcxproj.user
@@ -17,7 +17,7 @@
WindowsLocalDebugger
+set r_fullscreen 0 +set r_mode 9 +set sv_pure 0 +set fs_game e3
D:\projects\Doom3
- +set r_fullscreen 0 +set r_mode 9|+set r_fullscreen 1 +set r_mode 9 +set sv_pure 0 +set fs_game e3|+set r_fullscreen 0 +set r_mode 9 +set sv_pure 0|+set r_fullscreen 0 +set r_mode 9 +set sv_pure 0 |+set r_fullscreen 0 +set r_mode 9 +set sv_pure 0 +set fs_game e3|
+ +set r_fullscreen 0 +set r_mode 9|+set r_fullscreen 1 +set r_mode 9 +set sv_pure 0 +set fs_game e3|+set r_fullscreen 0 +set r_mode 9 +set sv_pure 0 |+set r_fullscreen 0 +set r_mode 9 +set sv_pure 0|+set r_fullscreen 0 +set r_mode 9 +set sv_pure 0 +set fs_game e3|
D:\projects\Doom3\Quake4.exe
diff --git a/neo/engine/opengl/gl_d3d12shim.cpp b/neo/engine/opengl/gl_d3d12shim.cpp
index e97d9c46..090b760f 100644
--- a/neo/engine/opengl/gl_d3d12shim.cpp
+++ b/neo/engine/opengl/gl_d3d12shim.cpp
@@ -586,20 +586,20 @@ struct NeuralPOMResource
uint64_t generation = 0;
uint64_t gpuGeneration = 0;
- std::vector weightsBytes;
- std::vector latentRGBA16FBytes;
-
- ComPtr weightsBuffer;
- ComPtr latentBuffer;
D3D12_RESOURCE_STATES weightsState = D3D12_RESOURCE_STATE_COMMON;
D3D12_RESOURCE_STATES latentState = D3D12_RESOURCE_STATE_COMMON;
+ UINT weightsSrvIndex = UINT_MAX;
+ UINT latentSrvIndex = UINT_MAX;
D3D12_CPU_DESCRIPTOR_HANDLE weightsSrvCpu{};
D3D12_GPU_DESCRIPTOR_HANDLE weightsSrvGpu{};
- UINT weightsSrvIndex = UINT_MAX;
D3D12_CPU_DESCRIPTOR_HANDLE latentSrvCpu{};
D3D12_GPU_DESCRIPTOR_HANDLE latentSrvGpu{};
- UINT latentSrvIndex = UINT_MAX;
+ ComPtr weightsBuffer;
+ ComPtr latentBuffer;
+
+ std::vector weightsBytes;
+ std::vector latentRGBA16FBytes;
};
struct TextureResource
@@ -607,22 +607,31 @@ struct TextureResource
GLuint glId = 0;
int width = 0;
int height = 0;
+ UINT srvIndex = UINT_MAX;
+ UINT mipLevels = 1;
+
GLenum format = GL_RGBA;
GLenum minFilter = GL_LINEAR;
GLenum magFilter = GL_LINEAR;
GLenum wrapS = GL_REPEAT;
GLenum wrapT = GL_REPEAT;
- std::vector sysmem;
-
- bool compressed = false;
GLenum compressedInternalFormat = 0;
UINT compressedBlockBytes = 0;
UINT compressedImageSize = 0;
- bool forceOpaqueAlpha = false;
-
DXGI_FORMAT dxgiFormat = DXGI_FORMAT_R8G8B8A8_UNORM;
- UINT mipLevels = 1;
+
+ bool compressed = false;
+ bool forceOpaqueAlpha = false;
+ bool gpuValid = false;
+ bool isNormalMap = false;
+ bool isGlowMap = false;
+ bool isSpecularMap = false;
+
+ ComPtr texture;
+ D3D12_RESOURCE_STATES state = D3D12_RESOURCE_STATE_COPY_DEST;
+ D3D12_CPU_DESCRIPTOR_HANDLE srvCpu{};
+ D3D12_GPU_DESCRIPTOR_HANDLE srvGpu{};
// Async mip generation state. Mip 0 is uploaded immediately; the CPU-built
// lower mips are produced by worker threads and copied into the D3D12 texture
@@ -633,23 +642,10 @@ struct TextureResource
uint64_t pendingMipGeneration = 0;
bool mipChainRequested = false;
bool mipChainResident = false;
+
+ std::vector sysmem;
std::vector> pendingMipChain;
- ComPtr texture;
- D3D12_CPU_DESCRIPTOR_HANDLE srvCpu{};
- D3D12_GPU_DESCRIPTOR_HANDLE srvGpu{};
- UINT srvIndex = UINT_MAX;
- bool gpuValid = false;
- D3D12_RESOURCE_STATES state = D3D12_RESOURCE_STATE_COPY_DEST;
-
- // Optional material role tags. Tagged textures are not treated as diffuse
- // inputs by default; they are selected as tangent-space normal/glow maps for
- // the fixed-function G-buffer path when explicitly bound or auto-discovered
- // from any active texture unit.
- bool isNormalMap = false;
- bool isGlowMap = false;
- bool isSpecularMap = false;
-
// Optional trained neural replacement for shader POM, usually attached to the
// normal-map texture. If this is absent or not resident, the old path runs.
NeuralPOMResource neuralPOM;
@@ -783,6 +779,8 @@ static float MapTexEnvMode(GLenum mode)
struct BatchKey
{
+ uint64_t fullHash = 0;
+
D3D12_VIEWPORT viewport;
D3D12_RECT scissor;
@@ -915,7 +913,128 @@ static bool TextureSrvArrayEquals(const UINT* a, const UINT* b)
return true;
}
-static bool BatchKeyEquals(const BatchKey& a, const BatchKey& b)
+static inline void QD3D12_MixHash(uint64_t& h, uint64_t v)
+{
+ h ^= v;
+ h *= 1099511628211ull;
+}
+
+static inline void QD3D12_MixHashFloat(uint64_t& h, float f)
+{
+ uint32_t bits = 0;
+ memcpy(&bits, &f, sizeof(bits));
+ QD3D12_MixHash(h, bits);
+}
+
+static inline void QD3D12_MixHashBytes(uint64_t& h, const void* data, size_t size)
+{
+ const uint8_t* bytes = static_cast(data);
+ for (size_t i = 0; i < size; ++i)
+ QD3D12_MixHash(h, bytes[i]);
+}
+
+static uint64_t QD3D12_HashBatchKey(const BatchKey& key)
+{
+ uint64_t h = 1469598103934665603ull;
+
+ QD3D12_MixHashBytes(h, &key.viewport, sizeof(key.viewport));
+ QD3D12_MixHashBytes(h, &key.scissor, sizeof(key.scissor));
+ QD3D12_MixHash(h, uint32_t(key.pipeline));
+ QD3D12_MixHash(h, uint32_t(key.topology));
+ QD3D12_MixHash(h, key.useTessellation ? 1ull : 0ull);
+ QD3D12_MixHash(h, key.tex0SrvIndex);
+ QD3D12_MixHash(h, key.tex1SrvIndex);
+ QD3D12_MixHashBytes(h, key.textureSrvIndex, sizeof(key.textureSrvIndex));
+ QD3D12_MixHash(h, key.normalMapSrvIndex);
+ QD3D12_MixHashFloat(h, key.useNormalMap);
+ QD3D12_MixHashFloat(h, key.normalMapStrength);
+ QD3D12_MixHashFloat(h, key.normalMapYSign);
+ QD3D12_MixHash(h, key.glowMapSrvIndex);
+ QD3D12_MixHashFloat(h, key.useGlowMap);
+ QD3D12_MixHashFloat(h, key.glowMapStrength);
+ QD3D12_MixHash(h, key.specularMapSrvIndex);
+ QD3D12_MixHashFloat(h, key.useSpecularMap);
+ QD3D12_MixHashFloat(h, key.specularMapStrength);
+ QD3D12_MixHash(h, key.neuralPOMWeightsSrvIndex);
+ QD3D12_MixHash(h, key.neuralPOMLatentSrvIndex);
+ QD3D12_MixHashFloat(h, key.useNeuralPOM);
+ QD3D12_MixHash(h, key.useARBPrograms ? 1ull : 0ull);
+ QD3D12_MixHash(h, key.arbVertexProgram);
+ QD3D12_MixHash(h, key.arbFragmentProgram);
+ QD3D12_MixHash(h, key.arbVertexRevision);
+ QD3D12_MixHash(h, key.arbFragmentRevision);
+ QD3D12_MixHash(h, reinterpret_cast(key.arbVertexBlob));
+ QD3D12_MixHash(h, reinterpret_cast(key.arbFragmentBlob));
+ if (key.useARBPrograms)
+ QD3D12_MixHashBytes(h, &key.arbConstants, sizeof(key.arbConstants));
+ QD3D12_MixHashBytes(h, key.texComb0RGB, sizeof(key.texComb0RGB));
+ QD3D12_MixHashBytes(h, key.texComb0Alpha, sizeof(key.texComb0Alpha));
+ QD3D12_MixHashBytes(h, key.texComb0Operand, sizeof(key.texComb0Operand));
+ QD3D12_MixHashBytes(h, key.texEnvColor0, sizeof(key.texEnvColor0));
+ QD3D12_MixHashBytes(h, key.texComb1RGB, sizeof(key.texComb1RGB));
+ QD3D12_MixHashBytes(h, key.texComb1Alpha, sizeof(key.texComb1Alpha));
+ QD3D12_MixHashBytes(h, key.texComb1Operand, sizeof(key.texComb1Operand));
+ QD3D12_MixHashBytes(h, key.texEnvColor1, sizeof(key.texEnvColor1));
+ QD3D12_MixHash(h, key.colorWriteMask);
+ QD3D12_MixHash(h, key.cullFaceEnabled ? 1ull : 0ull);
+ QD3D12_MixHash(h, uint32_t(key.cullMode));
+ QD3D12_MixHash(h, uint32_t(key.frontFace));
+ QD3D12_MixHash(h, key.stencilTest ? 1ull : 0ull);
+ QD3D12_MixHash(h, key.stencilReadMask);
+ QD3D12_MixHash(h, key.stencilWriteMask);
+ QD3D12_MixHash(h, key.stencilRef);
+ QD3D12_MixHash(h, uint32_t(key.stencilFrontFunc));
+ QD3D12_MixHash(h, uint32_t(key.stencilFrontSFail));
+ QD3D12_MixHash(h, uint32_t(key.stencilFrontDPFail));
+ QD3D12_MixHash(h, uint32_t(key.stencilFrontDPPass));
+ QD3D12_MixHash(h, uint32_t(key.stencilBackFunc));
+ QD3D12_MixHash(h, uint32_t(key.stencilBackSFail));
+ QD3D12_MixHash(h, uint32_t(key.stencilBackDPFail));
+ QD3D12_MixHash(h, uint32_t(key.stencilBackDPPass));
+ QD3D12_MixHash(h, key.depthBoundsTest ? 1ull : 0ull);
+ QD3D12_MixHashFloat(h, key.depthBoundsMin);
+ QD3D12_MixHashFloat(h, key.depthBoundsMax);
+ QD3D12_MixHash(h, key.polygonOffsetEnabled ? 1ull : 0ull);
+ QD3D12_MixHashFloat(h, key.polygonOffsetFactor);
+ QD3D12_MixHashFloat(h, key.polygonOffsetUnits);
+ QD3D12_MixHashFloat(h, key.alphaRef);
+ QD3D12_MixHashFloat(h, key.alphaFunc);
+ QD3D12_MixHashFloat(h, key.useTex0);
+ QD3D12_MixHashFloat(h, key.useTex1);
+ QD3D12_MixHashFloat(h, key.tex1IsLightmap);
+ QD3D12_MixHashFloat(h, key.texEnvMode0);
+ QD3D12_MixHashFloat(h, key.texEnvMode1);
+ QD3D12_MixHashFloat(h, key.fogEnabled);
+ QD3D12_MixHashFloat(h, key.fogMode);
+ QD3D12_MixHashFloat(h, key.fogDensity);
+ QD3D12_MixHashFloat(h, key.fogStart);
+ QD3D12_MixHashFloat(h, key.fogEnd);
+ QD3D12_MixHashBytes(h, key.fogColor, sizeof(key.fogColor));
+ QD3D12_MixHashBytes(h, key.currentColor, sizeof(key.currentColor));
+ QD3D12_MixHashFloat(h, key.useVertexColor);
+ QD3D12_MixHash(h, uint32_t(key.blendSrc));
+ QD3D12_MixHash(h, uint32_t(key.blendDst));
+ QD3D12_MixHash(h, key.depthTest ? 1ull : 0ull);
+ QD3D12_MixHash(h, key.depthWrite ? 1ull : 0ull);
+ QD3D12_MixHash(h, uint32_t(key.depthFunc));
+ QD3D12_MixHashBytes(h, key.mvp.m, sizeof(key.mvp.m));
+ QD3D12_MixHashBytes(h, key.prevMvp.m, sizeof(key.prevMvp.m));
+ QD3D12_MixHashBytes(h, key.modelMatrix.m, sizeof(key.modelMatrix.m));
+ QD3D12_MixHashFloat(h, key.geometryFlag);
+ QD3D12_MixHashFloat(h, key.roughness);
+ QD3D12_MixHashFloat(h, key.materialType);
+ QD3D12_MixHash(h, key.motionObjectId);
+ QD3D12_MixHashBytes(h, key.cameraWorldPos, sizeof(key.cameraWorldPos));
+ QD3D12_MixHashFloat(h, key.cameraValid);
+ return h ? h : 1ull;
+}
+
+static inline void QD3D12_FinalizeBatchKeyHash(BatchKey& key)
+{
+ key.fullHash = QD3D12_HashBatchKey(key);
+}
+
+static bool BatchKeyDeepEquals(const BatchKey& a, const BatchKey& b)
{
return
ViewportEquals(a.viewport, b.viewport) &&
@@ -1008,21 +1127,34 @@ static bool BatchKeyEquals(const BatchKey& a, const BatchKey& b)
memcmp(a.modelMatrix.m, b.modelMatrix.m, sizeof(a.modelMatrix.m)) == 0;
}
+static bool BatchKeyEquals(const BatchKey& a, const BatchKey& b)
+{
+ if (a.fullHash != 0 && b.fullHash != 0)
+ {
+#if defined(_DEBUG)
+ if (a.fullHash == b.fullHash)
+ assert(BatchKeyDeepEquals(a, b));
+#endif
+ return a.fullHash == b.fullHash;
+ }
+
+ return BatchKeyDeepEquals(a, b);
+}
+
struct QueuedBatch
{
- BatchKey key;
+ size_t firstVertex;
+ size_t vertexCount;
size_t markerBegin = 0;
size_t markerEnd = 0;
- size_t firstVertex;
- size_t vertexCount;
-
+ UINT gpuIndexCount = 0;
bool gpuIndexed = false;
D3D12_VERTEX_BUFFER_VIEW gpuVbv{};
D3D12_INDEX_BUFFER_VIEW gpuIbv{};
- UINT gpuIndexCount = 0;
ComPtr gpuVertexResource;
ComPtr gpuIndexResource;
+ BatchKey key;
};
struct QD3D12Window
@@ -1564,6 +1696,19 @@ struct GLState
GLState()
{
+ queries.max_load_factor(0.70f);
+ buffers.max_load_factor(0.70f);
+ prevObjectMVPs.max_load_factor(0.70f);
+ currObjectMVPs.max_load_factor(0.70f);
+ textures.max_load_factor(0.70f);
+ queries.reserve(QD3D12_MaxQueries);
+ queryMarkers.reserve(256);
+ buffers.reserve(256);
+ queuedBatches.reserve(4096);
+ prevObjectMVPs.reserve(1024);
+ currObjectMVPs.reserve(1024);
+ textures.reserve(QD3D12_MaxTextures);
+
for (UINT i = 0; i < QD3D12_MaxTextureUnits; ++i)
{
texture2D[i] = (i == 0);
@@ -5222,6 +5367,7 @@ static BatchKey BuildCurrentBatchKey(GLenum originalMode, const TextureResource*
if (key.useARBPrograms)
key.useTessellation = false;
g_gl.currObjectMVPs[key.motionObjectId] = key.mvp;
+ QD3D12_FinalizeBatchKeyHash(key);
return key;
}
@@ -11160,7 +11306,10 @@ static QueuedBatch* QD3D12_PrepareImmediateBatch(GLenum mode, size_t n)
BatchKey key = BuildCurrentBatchKey(mode, boundTextures[0], boundTextures[1], boundTextures, normalMapTex, glowMapTex, specularMapTex, true);
if (key.useTessellation && !QD3D12_ImmediateVertexCountCanUseNormalMapTessellation(mode, n))
+ {
key.useTessellation = false;
+ QD3D12_FinalizeBatchKeyHash(key);
+ }
const size_t markerCursor = g_gl.queryMarkers.size();
if (!g_gl.queuedBatches.empty() &&
@@ -11276,7 +11425,10 @@ static void FlushImmediate(GLenum mode, const GLVertex* src, size_t n)
BatchKey key = BuildCurrentBatchKey(mode, tex0, tex1, boundTextures, normalMapTex, glowMapTex, specularMapTex, true);
if (key.useTessellation && !QD3D12_ImmediateVertexCountCanUseNormalMapTessellation(mode, n))
+ {
key.useTessellation = false;
+ QD3D12_FinalizeBatchKeyHash(key);
+ }
const size_t markerCursor = g_gl.queryMarkers.size();
QueuedBatch* batch = nullptr;
diff --git a/neo/engine/renderer/draw_dx.cpp b/neo/engine/renderer/draw_dx.cpp
index fd6ccf27..02fca405 100644
--- a/neo/engine/renderer/draw_dx.cpp
+++ b/neo/engine/renderer/draw_dx.cpp
@@ -230,7 +230,7 @@ void RB_DXDrawInteractions(void)
return;
}
- glFinish();
+// glFinish();
glLightScene(backEnd.viewDef->renderWorld->dxrWorldId);
glRaytracingLightingClearLights(false);
}