diff --git a/neo/engine/doomdll.vcxproj.user b/neo/engine/doomdll.vcxproj.user index 08538da5..e4f905db 100644 --- a/neo/engine/doomdll.vcxproj.user +++ b/neo/engine/doomdll.vcxproj.user @@ -17,7 +17,7 @@ WindowsLocalDebugger +set r_fullscreen 0 +set r_mode 9 +set sv_pure 0 +set fs_game e3 D:\projects\Doom3 - +set r_fullscreen 0 +set r_mode 9|+set r_fullscreen 1 +set r_mode 9 +set sv_pure 0 +set fs_game e3|+set r_fullscreen 0 +set r_mode 9 +set sv_pure 0|+set r_fullscreen 0 +set r_mode 9 +set sv_pure 0 |+set r_fullscreen 0 +set r_mode 9 +set sv_pure 0 +set fs_game e3| + +set r_fullscreen 0 +set r_mode 9|+set r_fullscreen 1 +set r_mode 9 +set sv_pure 0 +set fs_game e3|+set r_fullscreen 0 +set r_mode 9 +set sv_pure 0 |+set r_fullscreen 0 +set r_mode 9 +set sv_pure 0|+set r_fullscreen 0 +set r_mode 9 +set sv_pure 0 +set fs_game e3| D:\projects\Doom3\Quake4.exe diff --git a/neo/engine/opengl/gl_d3d12shim.cpp b/neo/engine/opengl/gl_d3d12shim.cpp index e97d9c46..090b760f 100644 --- a/neo/engine/opengl/gl_d3d12shim.cpp +++ b/neo/engine/opengl/gl_d3d12shim.cpp @@ -586,20 +586,20 @@ struct NeuralPOMResource uint64_t generation = 0; uint64_t gpuGeneration = 0; - std::vector weightsBytes; - std::vector latentRGBA16FBytes; - - ComPtr weightsBuffer; - ComPtr latentBuffer; D3D12_RESOURCE_STATES weightsState = D3D12_RESOURCE_STATE_COMMON; D3D12_RESOURCE_STATES latentState = D3D12_RESOURCE_STATE_COMMON; + UINT weightsSrvIndex = UINT_MAX; + UINT latentSrvIndex = UINT_MAX; D3D12_CPU_DESCRIPTOR_HANDLE weightsSrvCpu{}; D3D12_GPU_DESCRIPTOR_HANDLE weightsSrvGpu{}; - UINT weightsSrvIndex = UINT_MAX; D3D12_CPU_DESCRIPTOR_HANDLE latentSrvCpu{}; D3D12_GPU_DESCRIPTOR_HANDLE latentSrvGpu{}; - UINT latentSrvIndex = UINT_MAX; + ComPtr weightsBuffer; + ComPtr latentBuffer; + + std::vector weightsBytes; + std::vector latentRGBA16FBytes; }; struct TextureResource @@ -607,22 +607,31 @@ struct TextureResource GLuint glId = 0; int width = 0; int height = 0; + UINT srvIndex = UINT_MAX; + UINT mipLevels = 1; + GLenum format = GL_RGBA; GLenum minFilter = GL_LINEAR; GLenum magFilter = GL_LINEAR; GLenum wrapS = GL_REPEAT; GLenum wrapT = GL_REPEAT; - std::vector sysmem; - - bool compressed = false; GLenum compressedInternalFormat = 0; UINT compressedBlockBytes = 0; UINT compressedImageSize = 0; - bool forceOpaqueAlpha = false; - DXGI_FORMAT dxgiFormat = DXGI_FORMAT_R8G8B8A8_UNORM; - UINT mipLevels = 1; + + bool compressed = false; + bool forceOpaqueAlpha = false; + bool gpuValid = false; + bool isNormalMap = false; + bool isGlowMap = false; + bool isSpecularMap = false; + + ComPtr texture; + D3D12_RESOURCE_STATES state = D3D12_RESOURCE_STATE_COPY_DEST; + D3D12_CPU_DESCRIPTOR_HANDLE srvCpu{}; + D3D12_GPU_DESCRIPTOR_HANDLE srvGpu{}; // Async mip generation state. Mip 0 is uploaded immediately; the CPU-built // lower mips are produced by worker threads and copied into the D3D12 texture @@ -633,23 +642,10 @@ struct TextureResource uint64_t pendingMipGeneration = 0; bool mipChainRequested = false; bool mipChainResident = false; + + std::vector sysmem; std::vector> pendingMipChain; - ComPtr texture; - D3D12_CPU_DESCRIPTOR_HANDLE srvCpu{}; - D3D12_GPU_DESCRIPTOR_HANDLE srvGpu{}; - UINT srvIndex = UINT_MAX; - bool gpuValid = false; - D3D12_RESOURCE_STATES state = D3D12_RESOURCE_STATE_COPY_DEST; - - // Optional material role tags. Tagged textures are not treated as diffuse - // inputs by default; they are selected as tangent-space normal/glow maps for - // the fixed-function G-buffer path when explicitly bound or auto-discovered - // from any active texture unit. - bool isNormalMap = false; - bool isGlowMap = false; - bool isSpecularMap = false; - // Optional trained neural replacement for shader POM, usually attached to the // normal-map texture. If this is absent or not resident, the old path runs. NeuralPOMResource neuralPOM; @@ -783,6 +779,8 @@ static float MapTexEnvMode(GLenum mode) struct BatchKey { + uint64_t fullHash = 0; + D3D12_VIEWPORT viewport; D3D12_RECT scissor; @@ -915,7 +913,128 @@ static bool TextureSrvArrayEquals(const UINT* a, const UINT* b) return true; } -static bool BatchKeyEquals(const BatchKey& a, const BatchKey& b) +static inline void QD3D12_MixHash(uint64_t& h, uint64_t v) +{ + h ^= v; + h *= 1099511628211ull; +} + +static inline void QD3D12_MixHashFloat(uint64_t& h, float f) +{ + uint32_t bits = 0; + memcpy(&bits, &f, sizeof(bits)); + QD3D12_MixHash(h, bits); +} + +static inline void QD3D12_MixHashBytes(uint64_t& h, const void* data, size_t size) +{ + const uint8_t* bytes = static_cast(data); + for (size_t i = 0; i < size; ++i) + QD3D12_MixHash(h, bytes[i]); +} + +static uint64_t QD3D12_HashBatchKey(const BatchKey& key) +{ + uint64_t h = 1469598103934665603ull; + + QD3D12_MixHashBytes(h, &key.viewport, sizeof(key.viewport)); + QD3D12_MixHashBytes(h, &key.scissor, sizeof(key.scissor)); + QD3D12_MixHash(h, uint32_t(key.pipeline)); + QD3D12_MixHash(h, uint32_t(key.topology)); + QD3D12_MixHash(h, key.useTessellation ? 1ull : 0ull); + QD3D12_MixHash(h, key.tex0SrvIndex); + QD3D12_MixHash(h, key.tex1SrvIndex); + QD3D12_MixHashBytes(h, key.textureSrvIndex, sizeof(key.textureSrvIndex)); + QD3D12_MixHash(h, key.normalMapSrvIndex); + QD3D12_MixHashFloat(h, key.useNormalMap); + QD3D12_MixHashFloat(h, key.normalMapStrength); + QD3D12_MixHashFloat(h, key.normalMapYSign); + QD3D12_MixHash(h, key.glowMapSrvIndex); + QD3D12_MixHashFloat(h, key.useGlowMap); + QD3D12_MixHashFloat(h, key.glowMapStrength); + QD3D12_MixHash(h, key.specularMapSrvIndex); + QD3D12_MixHashFloat(h, key.useSpecularMap); + QD3D12_MixHashFloat(h, key.specularMapStrength); + QD3D12_MixHash(h, key.neuralPOMWeightsSrvIndex); + QD3D12_MixHash(h, key.neuralPOMLatentSrvIndex); + QD3D12_MixHashFloat(h, key.useNeuralPOM); + QD3D12_MixHash(h, key.useARBPrograms ? 1ull : 0ull); + QD3D12_MixHash(h, key.arbVertexProgram); + QD3D12_MixHash(h, key.arbFragmentProgram); + QD3D12_MixHash(h, key.arbVertexRevision); + QD3D12_MixHash(h, key.arbFragmentRevision); + QD3D12_MixHash(h, reinterpret_cast(key.arbVertexBlob)); + QD3D12_MixHash(h, reinterpret_cast(key.arbFragmentBlob)); + if (key.useARBPrograms) + QD3D12_MixHashBytes(h, &key.arbConstants, sizeof(key.arbConstants)); + QD3D12_MixHashBytes(h, key.texComb0RGB, sizeof(key.texComb0RGB)); + QD3D12_MixHashBytes(h, key.texComb0Alpha, sizeof(key.texComb0Alpha)); + QD3D12_MixHashBytes(h, key.texComb0Operand, sizeof(key.texComb0Operand)); + QD3D12_MixHashBytes(h, key.texEnvColor0, sizeof(key.texEnvColor0)); + QD3D12_MixHashBytes(h, key.texComb1RGB, sizeof(key.texComb1RGB)); + QD3D12_MixHashBytes(h, key.texComb1Alpha, sizeof(key.texComb1Alpha)); + QD3D12_MixHashBytes(h, key.texComb1Operand, sizeof(key.texComb1Operand)); + QD3D12_MixHashBytes(h, key.texEnvColor1, sizeof(key.texEnvColor1)); + QD3D12_MixHash(h, key.colorWriteMask); + QD3D12_MixHash(h, key.cullFaceEnabled ? 1ull : 0ull); + QD3D12_MixHash(h, uint32_t(key.cullMode)); + QD3D12_MixHash(h, uint32_t(key.frontFace)); + QD3D12_MixHash(h, key.stencilTest ? 1ull : 0ull); + QD3D12_MixHash(h, key.stencilReadMask); + QD3D12_MixHash(h, key.stencilWriteMask); + QD3D12_MixHash(h, key.stencilRef); + QD3D12_MixHash(h, uint32_t(key.stencilFrontFunc)); + QD3D12_MixHash(h, uint32_t(key.stencilFrontSFail)); + QD3D12_MixHash(h, uint32_t(key.stencilFrontDPFail)); + QD3D12_MixHash(h, uint32_t(key.stencilFrontDPPass)); + QD3D12_MixHash(h, uint32_t(key.stencilBackFunc)); + QD3D12_MixHash(h, uint32_t(key.stencilBackSFail)); + QD3D12_MixHash(h, uint32_t(key.stencilBackDPFail)); + QD3D12_MixHash(h, uint32_t(key.stencilBackDPPass)); + QD3D12_MixHash(h, key.depthBoundsTest ? 1ull : 0ull); + QD3D12_MixHashFloat(h, key.depthBoundsMin); + QD3D12_MixHashFloat(h, key.depthBoundsMax); + QD3D12_MixHash(h, key.polygonOffsetEnabled ? 1ull : 0ull); + QD3D12_MixHashFloat(h, key.polygonOffsetFactor); + QD3D12_MixHashFloat(h, key.polygonOffsetUnits); + QD3D12_MixHashFloat(h, key.alphaRef); + QD3D12_MixHashFloat(h, key.alphaFunc); + QD3D12_MixHashFloat(h, key.useTex0); + QD3D12_MixHashFloat(h, key.useTex1); + QD3D12_MixHashFloat(h, key.tex1IsLightmap); + QD3D12_MixHashFloat(h, key.texEnvMode0); + QD3D12_MixHashFloat(h, key.texEnvMode1); + QD3D12_MixHashFloat(h, key.fogEnabled); + QD3D12_MixHashFloat(h, key.fogMode); + QD3D12_MixHashFloat(h, key.fogDensity); + QD3D12_MixHashFloat(h, key.fogStart); + QD3D12_MixHashFloat(h, key.fogEnd); + QD3D12_MixHashBytes(h, key.fogColor, sizeof(key.fogColor)); + QD3D12_MixHashBytes(h, key.currentColor, sizeof(key.currentColor)); + QD3D12_MixHashFloat(h, key.useVertexColor); + QD3D12_MixHash(h, uint32_t(key.blendSrc)); + QD3D12_MixHash(h, uint32_t(key.blendDst)); + QD3D12_MixHash(h, key.depthTest ? 1ull : 0ull); + QD3D12_MixHash(h, key.depthWrite ? 1ull : 0ull); + QD3D12_MixHash(h, uint32_t(key.depthFunc)); + QD3D12_MixHashBytes(h, key.mvp.m, sizeof(key.mvp.m)); + QD3D12_MixHashBytes(h, key.prevMvp.m, sizeof(key.prevMvp.m)); + QD3D12_MixHashBytes(h, key.modelMatrix.m, sizeof(key.modelMatrix.m)); + QD3D12_MixHashFloat(h, key.geometryFlag); + QD3D12_MixHashFloat(h, key.roughness); + QD3D12_MixHashFloat(h, key.materialType); + QD3D12_MixHash(h, key.motionObjectId); + QD3D12_MixHashBytes(h, key.cameraWorldPos, sizeof(key.cameraWorldPos)); + QD3D12_MixHashFloat(h, key.cameraValid); + return h ? h : 1ull; +} + +static inline void QD3D12_FinalizeBatchKeyHash(BatchKey& key) +{ + key.fullHash = QD3D12_HashBatchKey(key); +} + +static bool BatchKeyDeepEquals(const BatchKey& a, const BatchKey& b) { return ViewportEquals(a.viewport, b.viewport) && @@ -1008,21 +1127,34 @@ static bool BatchKeyEquals(const BatchKey& a, const BatchKey& b) memcmp(a.modelMatrix.m, b.modelMatrix.m, sizeof(a.modelMatrix.m)) == 0; } +static bool BatchKeyEquals(const BatchKey& a, const BatchKey& b) +{ + if (a.fullHash != 0 && b.fullHash != 0) + { +#if defined(_DEBUG) + if (a.fullHash == b.fullHash) + assert(BatchKeyDeepEquals(a, b)); +#endif + return a.fullHash == b.fullHash; + } + + return BatchKeyDeepEquals(a, b); +} + struct QueuedBatch { - BatchKey key; + size_t firstVertex; + size_t vertexCount; size_t markerBegin = 0; size_t markerEnd = 0; - size_t firstVertex; - size_t vertexCount; - + UINT gpuIndexCount = 0; bool gpuIndexed = false; D3D12_VERTEX_BUFFER_VIEW gpuVbv{}; D3D12_INDEX_BUFFER_VIEW gpuIbv{}; - UINT gpuIndexCount = 0; ComPtr gpuVertexResource; ComPtr gpuIndexResource; + BatchKey key; }; struct QD3D12Window @@ -1564,6 +1696,19 @@ struct GLState GLState() { + queries.max_load_factor(0.70f); + buffers.max_load_factor(0.70f); + prevObjectMVPs.max_load_factor(0.70f); + currObjectMVPs.max_load_factor(0.70f); + textures.max_load_factor(0.70f); + queries.reserve(QD3D12_MaxQueries); + queryMarkers.reserve(256); + buffers.reserve(256); + queuedBatches.reserve(4096); + prevObjectMVPs.reserve(1024); + currObjectMVPs.reserve(1024); + textures.reserve(QD3D12_MaxTextures); + for (UINT i = 0; i < QD3D12_MaxTextureUnits; ++i) { texture2D[i] = (i == 0); @@ -5222,6 +5367,7 @@ static BatchKey BuildCurrentBatchKey(GLenum originalMode, const TextureResource* if (key.useARBPrograms) key.useTessellation = false; g_gl.currObjectMVPs[key.motionObjectId] = key.mvp; + QD3D12_FinalizeBatchKeyHash(key); return key; } @@ -11160,7 +11306,10 @@ static QueuedBatch* QD3D12_PrepareImmediateBatch(GLenum mode, size_t n) BatchKey key = BuildCurrentBatchKey(mode, boundTextures[0], boundTextures[1], boundTextures, normalMapTex, glowMapTex, specularMapTex, true); if (key.useTessellation && !QD3D12_ImmediateVertexCountCanUseNormalMapTessellation(mode, n)) + { key.useTessellation = false; + QD3D12_FinalizeBatchKeyHash(key); + } const size_t markerCursor = g_gl.queryMarkers.size(); if (!g_gl.queuedBatches.empty() && @@ -11276,7 +11425,10 @@ static void FlushImmediate(GLenum mode, const GLVertex* src, size_t n) BatchKey key = BuildCurrentBatchKey(mode, tex0, tex1, boundTextures, normalMapTex, glowMapTex, specularMapTex, true); if (key.useTessellation && !QD3D12_ImmediateVertexCountCanUseNormalMapTessellation(mode, n)) + { key.useTessellation = false; + QD3D12_FinalizeBatchKeyHash(key); + } const size_t markerCursor = g_gl.queryMarkers.size(); QueuedBatch* batch = nullptr; diff --git a/neo/engine/renderer/draw_dx.cpp b/neo/engine/renderer/draw_dx.cpp index fd6ccf27..02fca405 100644 --- a/neo/engine/renderer/draw_dx.cpp +++ b/neo/engine/renderer/draw_dx.cpp @@ -230,7 +230,7 @@ void RB_DXDrawInteractions(void) return; } - glFinish(); +// glFinish(); glLightScene(backEnd.viewDef->renderWorld->dxrWorldId); glRaytracingLightingClearLights(false); }