From 00bd6c96d37d93431441b0be8e4f2be619e6e868 Mon Sep 17 00:00:00 2001 From: Exzap <13877693+Exzap@users.noreply.github.com> Date: Sat, 19 Sep 2026 03:52:28 +0200 Subject: [PATCH] Latte+GX2: Improve GX2SetShaderModeEx and handle uniform mode Cemu was more lenient than the actual hardware and allowed both uniform registers and uniform blocks use at the same time. We now handle DX9_CONSTS set via GX2SetShaderMode correctly and will try to zero out the disabled uniform type --- src/Cafe/HW/Latte/Core/FetchShader.cpp | 3 +- src/Cafe/HW/Latte/Core/LatteBufferData.cpp | 127 ++++++++++-------- .../HW/Latte/Core/LatteCommandProcessor.cpp | 5 +- src/Cafe/HW/Latte/ISA/LatteReg.h | 86 +++++++++++- .../LatteDecompiler.cpp | 5 + src/Cafe/OS/libs/gx2/GX2.cpp | 1 - src/Cafe/OS/libs/gx2/GX2.h | 1 - src/Cafe/OS/libs/gx2/GX2_Shader.cpp | 101 +++++++++++++- src/Cafe/OS/libs/gx2/GX2_Shader.h | 25 ++-- src/Cafe/OS/libs/gx2/GX2_shader_legacy.cpp | 35 ----- 10 files changed, 271 insertions(+), 118 deletions(-) diff --git a/src/Cafe/HW/Latte/Core/FetchShader.cpp b/src/Cafe/HW/Latte/Core/FetchShader.cpp index c292c9b9..d5e8bdd5 100644 --- a/src/Cafe/HW/Latte/Core/FetchShader.cpp +++ b/src/Cafe/HW/Latte/Core/FetchShader.cpp @@ -423,6 +423,7 @@ LatteFetchShader* LatteShaderRecompiler_createFetchShader(LatteFetchShader::Cach } newFetchShader->bufferGroups.shrink_to_fit(); // calculate group information + cemu_assert(newFetchShader->bufferGroups.size() <= Latte::GPU_LIMITS::NUM_VERTEX_BUFFERS); for (auto& bufferGroup : newFetchShader->bufferGroups) { bufferGroup.hasVtxIndexAccess = false; @@ -453,8 +454,6 @@ LatteFetchShader* LatteShaderRecompiler_createFetchShader(LatteFetchShader::Cach { newFetchShader->m_isRegistered = true; } - - return newFetchShader; } diff --git a/src/Cafe/HW/Latte/Core/LatteBufferData.cpp b/src/Cafe/HW/Latte/Core/LatteBufferData.cpp index 04181114..538f35e3 100644 --- a/src/Cafe/HW/Latte/Core/LatteBufferData.cpp +++ b/src/Cafe/HW/Latte/Core/LatteBufferData.cpp @@ -71,69 +71,86 @@ void rectGenerate4thVertex(uint32be* output, uint32be* input0, uint32be* input1, bool LatteBufferCache_LoadRemappedUniforms(LatteDecompilerShader* shader, float* uniformData, bool aluConstDirty, uint32 uniformBufferDirtyMask) { bool hasChange = false; - uint32 shaderAluConst; - uint32 shaderUniformRegisterOffset; - - switch (shader->shaderType) + if (LatteGPUState.contextNew.SQ_CONFIG.get_DX9_CONSTS()) { - case LatteConst::ShaderType::Vertex: - shaderAluConst = 0x400; - shaderUniformRegisterOffset = mmSQ_VTX_UNIFORM_BLOCK_START; - break; - case LatteConst::ShaderType::Pixel: - shaderAluConst = 0; - shaderUniformRegisterOffset = mmSQ_PS_UNIFORM_BLOCK_START; - break; - case LatteConst::ShaderType::Geometry: - shaderAluConst = 0; // geometry shader has no ALU const - shaderUniformRegisterOffset = mmSQ_GS_UNIFORM_BLOCK_START; - break; - default: - UNREACHABLE; - } - - // sourced from uniform registers - if (aluConstDirty) - { - uint32* aluConstBase = LatteGPUState.contextRegister + mmSQ_ALU_CONSTANT0_0 + shaderAluConst; - for (auto it : shader->list_remappedUniformEntries_register) + // ALU const only + uint32 shaderAluConst; + switch (shader->shaderType) { - uint64* __restrict uniformRegData = (uint64*)(aluConstBase + it.indexOffset / 4); - uint64* __restrict regDest = (uint64*)((uint8*)uniformData + it.mappedIndexOffset); - regDest[0] = uniformRegData[0]; - regDest[1] = uniformRegData[1]; + case LatteConst::ShaderType::Vertex: + shaderAluConst = 0x400; + break; + case LatteConst::ShaderType::Pixel: + shaderAluConst = 0; + break; + case LatteConst::ShaderType::Geometry: + shaderAluConst = 0; // geometry shader has no ALU const + break; + default: + UNREACHABLE; } - if (!shader->list_remappedUniformEntries_register.empty()) - hasChange = true; - } - // sourced from uniform buffers - if (uniformBufferDirtyMask) - { - for (auto& bufferGroup : shader->list_remappedUniformEntries_bufferGroups) + // sourced from uniform registers + if (aluConstDirty) { - if ((uniformBufferDirtyMask&(1<list_remappedUniformEntries_register) { - uint8* __restrict uniformBase = memory_base + physicalAddr; - for (auto& it : bufferGroup.entries) - { - uint64* __restrict regDest = (uint64*)((uint8*)uniformData + it.mappedIndexOffset); - uint64* __restrict uniformEntrySrc = (uint64*)(uniformBase + it.indexOffset); - memcpy(regDest, uniformEntrySrc, 16); - } + uint64* __restrict uniformRegData = (uint64*)(aluConstBase + it.indexOffset / 4); + uint64* __restrict regDest = (uint64*)((uint8*)uniformData + it.mappedIndexOffset); + regDest[0] = uniformRegData[0]; + regDest[1] = uniformRegData[1]; } - else + if (!shader->list_remappedUniformEntries_register.empty()) + hasChange = true; + } + } + else + { + // uniform blocks only + uint32 shaderUniformRegisterOffset; + switch (shader->shaderType) + { + case LatteConst::ShaderType::Vertex: + shaderUniformRegisterOffset = mmSQ_VTX_UNIFORM_BLOCK_START; + break; + case LatteConst::ShaderType::Pixel: + shaderUniformRegisterOffset = mmSQ_PS_UNIFORM_BLOCK_START; + break; + case LatteConst::ShaderType::Geometry: + shaderUniformRegisterOffset = mmSQ_GS_UNIFORM_BLOCK_START; + break; + default: + UNREACHABLE; + } + // sourced from uniform buffers + if (uniformBufferDirtyMask) + { + for (auto& bufferGroup : shader->list_remappedUniformEntries_bufferGroups) { - for (auto& it : bufferGroup.entries) + if ((uniformBufferDirtyMask&(1<bufferGroups.size() < 32); // fetch shader generation should guarantee + cemu_assert_debug(parsedFetchShader->bufferGroups.size() < Latte::GPU_LIMITS::NUM_VERTEX_BUFFERS); for (auto& bufferGroup : parsedFetchShader->bufferGroups) { uint32 bufferIndex = bufferGroup.attributeBufferIndex; diff --git a/src/Cafe/HW/Latte/Core/LatteCommandProcessor.cpp b/src/Cafe/HW/Latte/Core/LatteCommandProcessor.cpp index 5a4f1ce5..de677720 100644 --- a/src/Cafe/HW/Latte/Core/LatteCommandProcessor.cpp +++ b/src/Cafe/HW/Latte/Core/LatteCommandProcessor.cpp @@ -743,7 +743,8 @@ LatteCMDPtr LatteCP_itDrawIndexAuto(LatteCMDPtr cmd, uint32 nWords, DrawPassCont uint32 ukn = LatteReadCMD(); LatteGPUState.currentDrawCallTick = GetTickCount(); // todo - better way to identify compute drawcalls - if ((LatteGPUState.contextRegister[mmSQ_CONFIG] >> 24) == 0xE4) + auto& sqConfig = LatteGPUState.contextNew.SQ_CONFIG; + if (sqConfig.get_PS_PRIO() == 0 && sqConfig.get_VS_PRIO() == 1 && sqConfig.get_GS_PRIO() == 2 && sqConfig.get_ES_PRIO() == 3) { uint32 vsProgramCode = ((LatteGPUState.contextRegister[mmSQ_PGM_START_ES] & 0xFFFFFF) << 8); uint32 vsProgramSize = LatteGPUState.contextRegister[mmSQ_PGM_START_ES + 1] << 3; @@ -1199,6 +1200,7 @@ void LatteCP_processCommandBuffer(DrawPassContext& drawPassCtx) switch (itCode) { case IT_SET_CONTEXT_REG: + case IT_SET_ALL_CONTEXTS: { LatteCP_itSetRegistersGeneric(cmdData, nWords); } @@ -1484,6 +1486,7 @@ void LatteCP_ProcessRingbuffer() } break; case IT_SET_CONTEXT_REG: + case IT_SET_ALL_CONTEXTS: { LatteCP_itSetRegistersGeneric(cmd, nWords); timerRecheck += CP_TIMER_RECHECK / 512; diff --git a/src/Cafe/HW/Latte/ISA/LatteReg.h b/src/Cafe/HW/Latte/ISA/LatteReg.h index 6de723f0..5077ee73 100644 --- a/src/Cafe/HW/Latte/ISA/LatteReg.h +++ b/src/Cafe/HW/Latte/ISA/LatteReg.h @@ -398,6 +398,17 @@ namespace Latte { VGT_PRIMITIVE_TYPE = 0x2256, + SQ_CONFIG = 0x2300, + SQ_GPR_RESOURCE_MGMT_1 = 0x2301, + SQ_GPR_RESOURCE_MGMT_2 = 0x2302, + SQ_THREAD_RESOURCE_MGMT = 0x2303, + SQ_STACK_RESOURCE_MGMT_1 = 0x2304, + SQ_STACK_RESOURCE_MGMT_2 = 0x2305, + SQ_ESGS_RING_BASE = 0x2310, + SQ_ESGS_RING_SIZE = 0x2311, + SQ_GSVS_RING_BASE = 0x2312, + SQ_GSVS_RING_SIZE = 0x2313, + // each stage has 12 sets of 4 border color registers TD_PS_SAMPLER0_BORDER_RED = 0x2900, TD_PS_SAMPLER0_BORDER_GREEN = 0x2901, @@ -445,6 +456,7 @@ namespace Latte SPI_VS_OUT_ID_0 = 0xA185, SPI_VS_OUT_CONFIG = 0xA1B1, + SPI_THREAD_GROUPING = 0xA1B2, CB_BLEND0_CONTROL = 0xA1E0, // first CB_BLEND7_CONTROL = 0xA1E7, // last @@ -470,6 +482,8 @@ namespace Latte SQ_PGM_RESOURCES_ES = 0xA224, SQ_PGM_START_FS = 0xA225, SQ_PGM_RESOURCES_FS = 0xA229, + SQ_ESGS_RING_ITEMSIZE = 0xA22A, + SQ_GSVS_RING_ITEMSIZE = 0xA22B, SQ_VTX_SEMANTIC_CLEAR = 0xA238, @@ -487,6 +501,7 @@ namespace Latte VGT_INSTANCE_STEP_RATE_0 = 0xA2A8, VGT_INSTANCE_STEP_RATE_1 = 0xA2A9, + VGT_STRMOUT_EN = 0xA2AC, VGT_STRMOUT_BUFFER_SIZE_0 = 0xA2B4, VGT_STRMOUT_VTX_STRIDE_0 = 0xA2B5, @@ -634,6 +649,42 @@ float get_##__regname() const \ uint32 v{}; }; + struct LATTE_SQ_CONFIG : LATTEREG // 0x2300 + { + // todo: bits 0-1 + LATTE_BITFIELD_BOOL(DX9_CONSTS, 2); + LATTE_BITFIELD_BOOL(ALU_INST_PREFER_VECTOR, 3); + // todo: bits 4-23 + LATTE_BITFIELD(PS_PRIO, 24, 2); + LATTE_BITFIELD(VS_PRIO, 26, 2); + LATTE_BITFIELD(GS_PRIO, 28, 2); + LATTE_BITFIELD(ES_PRIO, 30, 2); + }; + + struct LATTE_SQ_GPR_RESOURCE_MGMT_1 : LATTEREG // 0x2301 + { + LATTE_BITFIELD(NUM_PS_GPRS, 0, 8); + LATTE_BITFIELD(NUM_VS_GPRS, 8, 8); + }; + + struct LATTE_SQ_GPR_RESOURCE_MGMT_2 : LATTEREG // 0x2302 + { + LATTE_BITFIELD(NUM_GS_GPRS, 0, 8); + LATTE_BITFIELD(NUM_ES_GPRS, 8, 8); + }; + + struct LATTE_SQ_STACK_RESOURCE_MGMT_1 : LATTEREG // 0x2304 + { + LATTE_BITFIELD(NUM_PS_STACK_ENTRIES, 0, 12); + LATTE_BITFIELD(NUM_VS_STACK_ENTRIES, 16, 12); + }; + + struct LATTE_SQ_STACK_RESOURCE_MGMT_2 : LATTEREG // 0x2305 + { + LATTE_BITFIELD(NUM_GS_STACK_ENTRIES, 0, 12); + LATTE_BITFIELD(NUM_ES_STACK_ENTRIES, 16, 12); + }; + // shared enums enum class E_COMPAREFUNC // used by depth test func and alpha test func { @@ -1481,7 +1532,19 @@ struct LatteContextRegister uint8 padding0[0x08958]; /* +0x08958 */ Latte::LATTE_VGT_PRIMITIVE_TYPE VGT_PRIMITIVE_TYPE; - uint8 padding5[0x0A400 - 0x0895C]; + uint8 padding_0895C[0x08C00 - 0x0895C]; + /* +0x08C00 */ Latte::LATTE_SQ_CONFIG SQ_CONFIG; + /* +0x08C04 */ Latte::LATTE_SQ_GPR_RESOURCE_MGMT_1 SQ_GPR_RESOURCE_MGMT_1; + /* +0x08C08 */ Latte::LATTE_SQ_GPR_RESOURCE_MGMT_2 SQ_GPR_RESOURCE_MGMT_2; + /* +0x08C0C */ uint32 SQ_THREAD_RESOURCE_MGMT; + /* +0x08C10 */ Latte::LATTE_SQ_STACK_RESOURCE_MGMT_1 SQ_STACK_RESOURCE_MGMT_1; + /* +0x08C14 */ Latte::LATTE_SQ_STACK_RESOURCE_MGMT_2 SQ_STACK_RESOURCE_MGMT_2; + uint8 padding_08C18[0x08C40 - 0x08C18]; + /* +0x08C40 */ uint32 SQ_ESGS_RING_BASE; + /* +0x08C44 */ uint32 SQ_ESGS_RING_SIZE; + /* +0x08C48 */ uint32 SQ_GSVS_RING_BASE; + /* +0x08C4C */ uint32 SQ_GSVS_RING_SIZE; + uint8 padding5[0x0A400 - 0x08C50]; /* +0x0A400 */ _LatteRegisterSetSamplerBorderColor TD_PS_SAMPLER_BORDER_COLOR[Latte::GPU_LIMITS::NUM_SAMPLERS_PER_STAGE]; uint8 padding6[0x0A600 - 0x0A520]; /* +0x0A600 */ _LatteRegisterSetSamplerBorderColor TD_VS_SAMPLER_BORDER_COLOR[Latte::GPU_LIMITS::NUM_SAMPLERS_PER_STAGE]; @@ -1520,7 +1583,8 @@ struct LatteContextRegister /* +0x286C4 */ Latte::LATTE_SPI_VS_OUT_CONFIG SPI_VS_OUT_CONFIG; - uint8 padding_286C8[0x28780 - 0x286C8]; + /* +0x286C8 */ uint32 SPI_THREAD_GROUPING; + uint8 padding_286CC[0x28780 - 0x286CC]; /* +0x28780 */ Latte::LATTE_CB_BLENDN_CONTROL CB_BLENDN_CONTROL[8]; @@ -1592,7 +1656,9 @@ struct LatteContextRegister /* +0x28AA0 */ Latte::LATTE_VGT_INSTANCE_STEP_RATE_X VGT_INSTANCE_STEP_RATE_0; /* +0x28AA4 */ Latte::LATTE_VGT_INSTANCE_STEP_RATE_X VGT_INSTANCE_STEP_RATE_1; - uint8 padding_28AA8[0x28AD0 - 0x28AA8]; + uint8 padding_28AA8[0x28AB0 - 0x28AA8]; + /* +0x28AB0 */ uint32 VGT_STRMOUT_EN; + uint8 padding_28AB4[0x28AD0 - 0x28AB4]; /* +0x28AD0 */ _LatteRegisterSetStreamoutBuffer VGT_STRMOUT_BUFFER_X[4]; /* +0x28B10 */ Latte::LATTE_VGT_STRMOUT_BASE_OFFSET_X VGT_STRMOUT_BASE_OFFSET_X[4]; @@ -1663,6 +1729,20 @@ struct LatteContextRegister static_assert(sizeof(LatteContextRegister) == 0x10000 * 4 + 9 * 4); static_assert(offsetof(LatteContextRegister, VGT_PRIMITIVE_TYPE) == Latte::REGADDR::VGT_PRIMITIVE_TYPE * 4); +static_assert(offsetof(LatteContextRegister, SQ_CONFIG) == Latte::REGADDR::SQ_CONFIG * 4); +static_assert(offsetof(LatteContextRegister, SQ_GPR_RESOURCE_MGMT_1) == Latte::REGADDR::SQ_GPR_RESOURCE_MGMT_1 * 4); +static_assert(offsetof(LatteContextRegister, SQ_GPR_RESOURCE_MGMT_2) == Latte::REGADDR::SQ_GPR_RESOURCE_MGMT_2 * 4); +static_assert(offsetof(LatteContextRegister, SQ_THREAD_RESOURCE_MGMT) == Latte::REGADDR::SQ_THREAD_RESOURCE_MGMT * 4); +static_assert(offsetof(LatteContextRegister, SQ_STACK_RESOURCE_MGMT_1) == Latte::REGADDR::SQ_STACK_RESOURCE_MGMT_1 * 4); +static_assert(offsetof(LatteContextRegister, SQ_STACK_RESOURCE_MGMT_2) == Latte::REGADDR::SQ_STACK_RESOURCE_MGMT_2 * 4); +static_assert(offsetof(LatteContextRegister, SQ_ESGS_RING_BASE) == Latte::REGADDR::SQ_ESGS_RING_BASE * 4); +static_assert(offsetof(LatteContextRegister, SQ_ESGS_RING_SIZE) == Latte::REGADDR::SQ_ESGS_RING_SIZE * 4); +static_assert(offsetof(LatteContextRegister, SQ_GSVS_RING_BASE) == Latte::REGADDR::SQ_GSVS_RING_BASE * 4); +static_assert(offsetof(LatteContextRegister, SQ_GSVS_RING_SIZE) == Latte::REGADDR::SQ_GSVS_RING_SIZE * 4); +static_assert(offsetof(LatteContextRegister, SPI_THREAD_GROUPING) == Latte::REGADDR::SPI_THREAD_GROUPING * 4); +static_assert(offsetof(LatteContextRegister, SQ_ESGS_RING_ITEMSIZE) == Latte::REGADDR::SQ_ESGS_RING_ITEMSIZE * 4); +static_assert(offsetof(LatteContextRegister, SQ_GSVS_RING_ITEMSIZE) == Latte::REGADDR::SQ_GSVS_RING_ITEMSIZE * 4); +static_assert(offsetof(LatteContextRegister, VGT_STRMOUT_EN) == Latte::REGADDR::VGT_STRMOUT_EN * 4); static_assert(offsetof(LatteContextRegister, TD_PS_SAMPLER_BORDER_COLOR) == Latte::REGADDR::TD_PS_SAMPLER0_BORDER_RED * 4); static_assert(offsetof(LatteContextRegister, TD_VS_SAMPLER_BORDER_COLOR) == Latte::REGADDR::TD_VS_SAMPLER0_BORDER_RED * 4); static_assert(offsetof(LatteContextRegister, TD_GS_SAMPLER_BORDER_COLOR) == Latte::REGADDR::TD_GS_SAMPLER0_BORDER_RED * 4); diff --git a/src/Cafe/HW/Latte/LegacyShaderDecompiler/LatteDecompiler.cpp b/src/Cafe/HW/Latte/LegacyShaderDecompiler/LatteDecompiler.cpp index 866d6818..6bccb944 100644 --- a/src/Cafe/HW/Latte/LegacyShaderDecompiler/LatteDecompiler.cpp +++ b/src/Cafe/HW/Latte/LegacyShaderDecompiler/LatteDecompiler.cpp @@ -1068,6 +1068,11 @@ void _LatteDecompiler_Process(LatteDecompilerShaderContext* shaderContext, uint8 LatteDecompiler_analyze(shaderContext, shaderContext->shader); if (shaderContext->shader->hasError == false) LatteDecompiler_analyzeDataTypes(shaderContext); + // check for usage errors + if ( shaderContext->analyzer.uniformRegisterAccessTracker.HasAccess() && shaderContext->analyzer.uniformBufferAccessTracker->HasAccess() ) + { + cemuLog_log(LogType::APIErrors, "Shader {:08x} accesses both uniform registers and uniform blocks. Latte does not support using both at the same time (uniform mode is configured via GX2SetShaderModeEx)", shaderContext->shaderBaseHash); + } // emit code if (shaderContext->shader->hasError == false) { diff --git a/src/Cafe/OS/libs/gx2/GX2.cpp b/src/Cafe/OS/libs/gx2/GX2.cpp index dcaa240b..248ae88d 100644 --- a/src/Cafe/OS/libs/gx2/GX2.cpp +++ b/src/Cafe/OS/libs/gx2/GX2.cpp @@ -338,7 +338,6 @@ namespace GX2 osLib_addFunction("gx2", "GX2SetPixelUniformBlock", gx2Export_GX2SetPixelUniformBlock); osLib_addFunction("gx2", "GX2SetGeometryUniformBlock", gx2Export_GX2SetGeometryUniformBlock); - osLib_addFunction("gx2", "GX2SetShaderModeEx", gx2Export_GX2SetShaderModeEx); osLib_addFunction("gx2", "GX2CalcGeometryShaderInputRingBufferSize", gx2Export_GX2CalcGeometryShaderInputRingBufferSize); osLib_addFunction("gx2", "GX2CalcGeometryShaderOutputRingBufferSize", gx2Export_GX2CalcGeometryShaderOutputRingBufferSize); diff --git a/src/Cafe/OS/libs/gx2/GX2.h b/src/Cafe/OS/libs/gx2/GX2.h index 168e0297..b722fca6 100644 --- a/src/Cafe/OS/libs/gx2/GX2.h +++ b/src/Cafe/OS/libs/gx2/GX2.h @@ -26,7 +26,6 @@ void gx2Export_GX2SetVertexUniformBlock(PPCInterpreter_t* hCPU); void gx2Export_GX2RSetVertexUniformBlock(PPCInterpreter_t* hCPU); void gx2Export_GX2SetPixelUniformBlock(PPCInterpreter_t* hCPU); void gx2Export_GX2SetGeometryUniformBlock(PPCInterpreter_t* hCPU); -void gx2Export_GX2SetShaderModeEx(PPCInterpreter_t* hCPU); void gx2Export_GX2CalcGeometryShaderInputRingBufferSize(PPCInterpreter_t* hCPU); void gx2Export_GX2CalcGeometryShaderOutputRingBufferSize(PPCInterpreter_t* hCPU); diff --git a/src/Cafe/OS/libs/gx2/GX2_Shader.cpp b/src/Cafe/OS/libs/gx2/GX2_Shader.cpp index bfd18f4b..b0ee2cb7 100644 --- a/src/Cafe/OS/libs/gx2/GX2_Shader.cpp +++ b/src/Cafe/OS/libs/gx2/GX2_Shader.cpp @@ -1,13 +1,13 @@ #include "Cafe/OS/common/OSCommon.h" #include "GX2.h" #include "GX2_Shader.h" +#include "GX2_Misc.h" +#include "Cafe/OS/libs/coreinit/coreinit_Misc.h" #include "Cafe/HW/Latte/Core/LatteConst.h" #include "Cafe/HW/Latte/Core/LattePM4.h" #include "Cafe/HW/Latte/ISA/LatteReg.h" #include "Cafe/HW/Latte/ISA/LatteInstructions.h" -uint32 memory_getVirtualOffsetFromPointer(void* ptr); // remove once we updated everything to MEMPTR - namespace GX2 { using namespace Latte; @@ -465,6 +465,99 @@ namespace GX2 _GX2SubmitUniformReg(0, offset, values, sizeInU32s); } + void GX2SetShaderModeEx(GX2_SHADER_MODE mode, uint32 shaderGprsVS, uint32 shaderStackVS, uint32 shaderGprsGS, uint32 shaderStackGS, uint32 shaderGprsPS, uint32 shaderStackPS) + { + if (static_cast(mode) > static_cast(GX2_SHADER_MODE::COMPUTE_SHADER)) + { + cemu_assert_suspicious(); + return; + } + bool isGeometry = mode == GX2_SHADER_MODE::GEOMETRY_SHADER; + bool isCompute = mode == GX2_SHADER_MODE::COMPUTE_SHADER; + bool setThreadGrouping = !isCompute && coreinit::__OSGetProcessSDKVersion() >= 21104; + GX2ReserveCmdSpace((isCompute ? 26 : isGeometry ? 8 : 11) + (setThreadGrouping ? 3 : 0)); + // geometry mode sets this in GX2SetGeometryShader + if (!isGeometry) + { + Latte::LATTE_VGT_GS_MODE gsMode; + if (isCompute) + { + gsMode.set_MODE(Latte::LATTE_VGT_GS_MODE::E_MODE::SCENARIO_G) + .set_COMPUTE_MODE(Latte::LATTE_VGT_GS_MODE::E_COMPUTE_MODE::ON) + .set_PARTIAL_THD_AT_EOI(true); + } + gx2WriteGather_submit(pm4HeaderType3(IT_SET_CONTEXT_REG, 2), + Latte::REGADDR::VGT_GS_MODE - LATTE_REG_BASE_CONTEXT, + gsMode); + } + + Latte::LATTE_SQ_CONFIG sqConfig; + sqConfig.set_DX9_CONSTS(mode == GX2_SHADER_MODE::UNIFORM_REGISTER) + .set_ALU_INST_PREFER_VECTOR(true) + .set_PS_PRIO(isCompute ? 0 : 3) + .set_VS_PRIO(isCompute ? 1 : 2) + .set_GS_PRIO(isCompute ? 2 : 1) + .set_ES_PRIO(isCompute ? 3 : 0); + uint32 threadResource = 0x04043088; + Latte::LATTE_SQ_GPR_RESOURCE_MGMT_1 gprResource1; + Latte::LATTE_SQ_GPR_RESOURCE_MGMT_2 gprResource2; + Latte::LATTE_SQ_STACK_RESOURCE_MGMT_1 stackResource1; + Latte::LATTE_SQ_STACK_RESOURCE_MGMT_2 stackResource2; + if (isCompute) + { + threadResource = 0xBD010101; + gprResource2.set_NUM_ES_GPRS(0xF8); + stackResource2.set_NUM_ES_STACK_ENTRIES(0x100); + } + else + { + gprResource1.set_NUM_PS_GPRS(shaderGprsPS); + stackResource1.set_NUM_PS_STACK_ENTRIES(shaderStackPS); + if (isGeometry) + { + // in geometry shader mode, the VS runs in ES and VS runs the geometry copy shader + gprResource1.set_NUM_VS_GPRS(0x40); + gprResource2.set_NUM_GS_GPRS(shaderGprsGS).set_NUM_ES_GPRS(shaderGprsVS); + stackResource2.set_NUM_GS_STACK_ENTRIES(shaderStackGS).set_NUM_ES_STACK_ENTRIES(shaderStackVS); + threadResource = 0x1C08207C; + } + else + { + gprResource1.set_NUM_VS_GPRS(shaderGprsVS); + stackResource1.set_NUM_VS_STACK_ENTRIES(shaderStackVS); + } + } + + gx2WriteGather_submit(pm4HeaderType3(IT_SET_CONFIG_REG, 7), + Latte::REGADDR::SQ_CONFIG - LATTE_REG_BASE_CONFIG, + sqConfig, gprResource1.getRawValue() | 0x40000000, + gprResource2, threadResource, stackResource1, stackResource2); + + if (isCompute) + { + gx2WriteGather_submit(pm4HeaderType3(IT_SET_CONFIG_REG, 5), + Latte::REGADDR::SQ_ESGS_RING_BASE - LATTE_REG_BASE_CONFIG, + 0, 0xFFFFFF, 0, 0xFFFFFF); + gx2WriteGather_submit(pm4HeaderType3(IT_SET_CONTEXT_REG, 2), + Latte::REGADDR::SQ_ESGS_RING_ITEMSIZE - LATTE_REG_BASE_CONTEXT, + 0); + gx2WriteGather_submit(pm4HeaderType3(IT_SET_CONTEXT_REG, 2), + Latte::REGADDR::SQ_GSVS_RING_ITEMSIZE - LATTE_REG_BASE_CONTEXT, + 1); + gx2WriteGather_submit(pm4HeaderType3(IT_SET_CONTEXT_REG, 2), + Latte::REGADDR::VGT_STRMOUT_EN - LATTE_REG_BASE_CONTEXT, + 0); + } + else if (setThreadGrouping) + { + gx2WriteGather_submit(pm4HeaderType3(IT_SET_ALL_CONTEXTS, 2), + Latte::REGADDR::SPI_THREAD_GROUPING - LATTE_REG_BASE_CONTEXT, + 1); + } + if (mode != GX2_SHADER_MODE::UNIFORM_REGISTER) + GX2Invalidate(GX2InvalidationFlag::GPU_SHADER, MPTR_NULL, 0xFFFFFFFF); + } + void GX2ShaderInit() { cafeExportRegister("gx2", GX2CalcFetchShaderSizeEx, LogType::GX2); @@ -479,5 +572,7 @@ namespace GX2 cafeExportRegister("gx2", GX2SetVertexUniformReg, LogType::GX2); cafeExportRegister("gx2", GX2SetPixelUniformReg, LogType::GX2); + + cafeExportRegister("gx2", GX2SetShaderModeEx, LogType::GX2); } -} \ No newline at end of file +} diff --git a/src/Cafe/OS/libs/gx2/GX2_Shader.h b/src/Cafe/OS/libs/gx2/GX2_Shader.h index 960bdf95..65f15e36 100644 --- a/src/Cafe/OS/libs/gx2/GX2_Shader.h +++ b/src/Cafe/OS/libs/gx2/GX2_Shader.h @@ -28,26 +28,17 @@ static_assert(sizeof(betype) == 4); namespace GX2 { + enum class GX2_SHADER_MODE : uint32 + { + UNIFORM_REGISTER = 0, + UNIFORM_BLOCK = 1, + GEOMETRY_SHADER = 2, + COMPUTE_SHADER = 3, + }; void GX2ShaderInit(); } -// code below still needs to be modernized (use betype, enum classes, move to namespace) - -// deprecated, use GX2_SHADER_MODE enum class instead -#define GX2_SHADER_MODE_UNIFORM_REGISTER 0 -#define GX2_SHADER_MODE_UNIFORM_BLOCK 1 -#define GX2_SHADER_MODE_GEOMETRY_SHADER 2 -#define GX2_SHADER_MODE_COMPUTE_SHADER 3 - -enum class GX2_SHADER_MODE : uint32 -{ - UNIFORM_REGISTER = 0, - UNIFORM_BLOCK = 1, - GEOMETRY_SHADER = 2, - COMPUTE_SHADER = 3, -}; - struct GX2VertexShader { /* +0x000 */ @@ -68,7 +59,7 @@ struct GX2VertexShader }regs; /* +0x0D0 */ uint32be shaderSize; /* +0x0D4 */ MEMPTR shaderPtr; - /* +0x0D8 */ betype shaderMode; + /* +0x0D8 */ betype shaderMode; /* +0x0DC */ uint32 uniformBlockCount; /* +0x0E0 */ MPTR uniformBlockInfo; /* +0x0E4 */ uint32 uniformVarCount; diff --git a/src/Cafe/OS/libs/gx2/GX2_shader_legacy.cpp b/src/Cafe/OS/libs/gx2/GX2_shader_legacy.cpp index d91a8529..560ae05a 100644 --- a/src/Cafe/OS/libs/gx2/GX2_shader_legacy.cpp +++ b/src/Cafe/OS/libs/gx2/GX2_shader_legacy.cpp @@ -326,41 +326,6 @@ void gx2Export_GX2RSetVertexUniformBlock(PPCInterpreter_t* hCPU) osLib_returnFromFunction(hCPU, 0); } -void gx2Export_GX2SetShaderModeEx(PPCInterpreter_t* hCPU) -{ - GX2::GX2ReserveCmdSpace(8+4); - uint32 mode = hCPU->gpr[3]; - - uint32 sqConfig = hCPU->gpr[3] == 0 ? 4 : 0; - if (mode == GX2_SHADER_MODE_COMPUTE_SHADER) - sqConfig |= 0xE4000000; // ES/GS/PS priority? - // todo - other sqConfig bits - - gx2WriteGather_submit((uint32)(pm4HeaderType3(IT_SET_CONFIG_REG, 7)), - (uint32)(mmSQ_CONFIG - 0x2000), - sqConfig, - 0, // ukn / todo - 0, // ukn / todo - 0, // ukn / todo - 0, // ukn / todo - 0 // ukn / todo - ); - - // if not GS, then update mmVGT_GS_MODE - if( mode != GX2_SHADER_MODE_GEOMETRY_SHADER ) - { - // update VGT_GS_MODE only if no geometry shader is used (else this register is already set by GX2SetGeometryShader) - gx2WriteGather_submitU32AsBE(pm4HeaderType3(IT_SET_CONTEXT_REG, 2)); - gx2WriteGather_submitU32AsBE(Latte::REGADDR::VGT_GS_MODE-0xA000); - if (mode == GX2_SHADER_MODE_COMPUTE_SHADER) - gx2WriteGather_submitU32AsBE(Latte::LATTE_VGT_GS_MODE().set_MODE(Latte::LATTE_VGT_GS_MODE::E_MODE::SCENARIO_G).set_COMPUTE_MODE(Latte::LATTE_VGT_GS_MODE::E_COMPUTE_MODE::ON).set_PARTIAL_THD_AT_EOI(true).getRawValueBE()); - else - gx2WriteGather_submitU32AsBE(_swapEndianU32(0)); - } - - osLib_returnFromFunction(hCPU, 0); -} - void gx2Export_GX2CalcGeometryShaderInputRingBufferSize(PPCInterpreter_t* hCPU) { uint32 size = (hCPU->gpr[3]*4) * 0x1000;