From f722bdf7f1ab5913d035f257ff82865a686b08d6 Mon Sep 17 00:00:00 2001 From: Pokechu22 Date: Wed, 13 Apr 2022 20:57:38 -0700 Subject: VertexLoaderX64: Refactor so that zfreeze is only in one place (Specifically, the copy for VertexLoaderManager::position_cache. The position matrix index happens elsewhere, and the float path still has special logic to copy to scratch3.) --- Source/Core/VideoCommon/VertexLoaderX64.cpp | 32 ++++++++++++----------------- 1 file changed, 13 insertions(+), 19 deletions(-) (limited to 'Source/Core/VideoCommon/VertexLoaderX64.cpp') diff --git a/Source/Core/VideoCommon/VertexLoaderX64.cpp b/Source/Core/VideoCommon/VertexLoaderX64.cpp index 7a4929361d..734fa0819c 100644 --- a/Source/Core/VideoCommon/VertexLoaderX64.cpp +++ b/Source/Core/VideoCommon/VertexLoaderX64.cpp @@ -114,6 +114,17 @@ int VertexLoaderX64::ReadVertex(OpArg data, VertexComponentFormat attribute, Com X64Reg coords = XMM0; + const auto write_zfreeze = [&]() { // zfreeze + if (native_format == &m_native_vtx_decl.position) + { + CMP(32, R(count_reg), Imm8(3)); + FixupBranch dont_store = J_CC(CC_A); + LEA(32, scratch3, MScaled(count_reg, SCALE_4, -4)); + MOVUPS(MPIC(VertexLoaderManager::position_cache, scratch3, SCALE_4), coords); + SetJumpTarget(dont_store); + } + }; + int elem_size = GetElementSize(format); int load_bytes = elem_size * count_in; OpArg dest = MDisp(dst_reg, m_dst_ofs); @@ -217,16 +228,7 @@ int VertexLoaderX64::ReadVertex(OpArg data, VertexComponentFormat attribute, Com } } - // zfreeze - if (native_format == &m_native_vtx_decl.position) - { - CMP(32, R(count_reg), Imm8(3)); - FixupBranch dont_store = J_CC(CC_A); - LEA(32, scratch3, MScaled(count_reg, SCALE_4, -4)); - MOVUPS(MPIC(VertexLoaderManager::position_cache, scratch3, SCALE_4), coords); - SetJumpTarget(dont_store); - } - return load_bytes; + write_zfreeze(); } } @@ -251,15 +253,7 @@ int VertexLoaderX64::ReadVertex(OpArg data, VertexComponentFormat attribute, Com break; } - // zfreeze - if (native_format == &m_native_vtx_decl.position) - { - CMP(32, R(count_reg), Imm8(3)); - FixupBranch dont_store = J_CC(CC_A); - LEA(32, scratch3, MScaled(count_reg, SCALE_4, -4)); - MOVUPS(MPIC(VertexLoaderManager::position_cache, scratch3, SCALE_4), coords); - SetJumpTarget(dont_store); - } + write_zfreeze(); return load_bytes; } -- cgit v1.2.3 From 97d0ff58c8dbe374a1b87e1a72f1f245ed27ed96 Mon Sep 17 00:00:00 2001 From: Pokechu22 Date: Wed, 13 Apr 2022 16:12:53 -0700 Subject: Convert vertex loader position cache to std::array --- Source/Core/VideoCommon/VertexLoaderX64.cpp | 8 +++++--- 1 file changed, 5 insertions(+), 3 deletions(-) (limited to 'Source/Core/VideoCommon/VertexLoaderX64.cpp') diff --git a/Source/Core/VideoCommon/VertexLoaderX64.cpp b/Source/Core/VideoCommon/VertexLoaderX64.cpp index 734fa0819c..91889c83d9 100644 --- a/Source/Core/VideoCommon/VertexLoaderX64.cpp +++ b/Source/Core/VideoCommon/VertexLoaderX64.cpp @@ -119,8 +119,9 @@ int VertexLoaderX64::ReadVertex(OpArg data, VertexComponentFormat attribute, Com { CMP(32, R(count_reg), Imm8(3)); FixupBranch dont_store = J_CC(CC_A); - LEA(32, scratch3, MScaled(count_reg, SCALE_4, -4)); - MOVUPS(MPIC(VertexLoaderManager::position_cache, scratch3, SCALE_4), coords); + LEA(32, scratch3, + MScaled(count_reg, SCALE_4, -int(VertexLoaderManager::position_cache[0].size()))); + MOVUPS(MPIC(VertexLoaderManager::position_cache.data(), scratch3, SCALE_4), coords); SetJumpTarget(dont_store); } }; @@ -408,7 +409,8 @@ void VertexLoaderX64::GenerateVertexLoader() // zfreeze CMP(32, R(count_reg), Imm8(3)); FixupBranch dont_store = J_CC(CC_A); - MOV(32, MPIC(VertexLoaderManager::position_matrix_index, count_reg, SCALE_4), R(scratch1)); + MOV(32, MPIC(VertexLoaderManager::position_matrix_index_cache.data(), count_reg, SCALE_4), + R(scratch1)); SetJumpTarget(dont_store); m_native_vtx_decl.posmtx.components = 4; -- cgit v1.2.3 From 39b2854b981aadec1576254ad5e4d91dc21c9709 Mon Sep 17 00:00:00 2001 From: Pokechu22 Date: Thu, 14 Apr 2022 12:01:57 -0700 Subject: VertexLoader: Convert count register to remaining register This more accurately represents what's going on, and also ends at 0 instead of 1, making some indexing operations easier. This also changes it so that position_matrix_index_cache actually starts from index 0 instead of index 1. --- Source/Core/VideoCommon/VertexLoaderX64.cpp | 31 +++++++++++++++++------------ 1 file changed, 18 insertions(+), 13 deletions(-) (limited to 'Source/Core/VideoCommon/VertexLoaderX64.cpp') diff --git a/Source/Core/VideoCommon/VertexLoaderX64.cpp b/Source/Core/VideoCommon/VertexLoaderX64.cpp index 91889c83d9..da52788d3c 100644 --- a/Source/Core/VideoCommon/VertexLoaderX64.cpp +++ b/Source/Core/VideoCommon/VertexLoaderX64.cpp @@ -26,7 +26,9 @@ static const X64Reg dst_reg = ABI_PARAM2; static const X64Reg scratch1 = RAX; static const X64Reg scratch2 = ABI_PARAM3; static const X64Reg scratch3 = ABI_PARAM4; -static const X64Reg count_reg = R10; +// The remaining number of vertices to be processed. Starts at count - 1, and the final loop has it +// at 0. +static const X64Reg remaining_reg = R10; static const X64Reg skipped_reg = R11; static const X64Reg base_reg = RBX; @@ -117,10 +119,11 @@ int VertexLoaderX64::ReadVertex(OpArg data, VertexComponentFormat attribute, Com const auto write_zfreeze = [&]() { // zfreeze if (native_format == &m_native_vtx_decl.position) { - CMP(32, R(count_reg), Imm8(3)); - FixupBranch dont_store = J_CC(CC_A); - LEA(32, scratch3, - MScaled(count_reg, SCALE_4, -int(VertexLoaderManager::position_cache[0].size()))); + CMP(32, R(remaining_reg), Imm8(3)); + FixupBranch dont_store = J_CC(CC_AE); + // The position cache is composed of 3 rows of 4 floats each; since each float is 4 bytes, + // we need to scale by 4 twice to cover the 4 floats. + LEA(32, scratch3, MScaled(remaining_reg, SCALE_4, 0)); MOVUPS(MPIC(VertexLoaderManager::position_cache.data(), scratch3, SCALE_4), coords); SetJumpTarget(dont_store); } @@ -380,8 +383,8 @@ void VertexLoaderX64::ReadColor(OpArg data, VertexComponentFormat attribute, Col void VertexLoaderX64::GenerateVertexLoader() { - BitSet32 regs = {src_reg, dst_reg, scratch1, scratch2, - scratch3, count_reg, skipped_reg, base_reg}; + BitSet32 regs = {src_reg, dst_reg, scratch1, scratch2, + scratch3, remaining_reg, skipped_reg, base_reg}; regs &= ABI_ALL_CALLEE_SAVED; ABI_PushRegistersAndAdjustStack(regs, 0); @@ -389,7 +392,9 @@ void VertexLoaderX64::GenerateVertexLoader() PUSH(32, R(ABI_PARAM3)); // ABI_PARAM3 is one of the lower registers, so free it for scratch2. - MOV(32, R(count_reg), R(ABI_PARAM3)); + // We also have it end at a value of 0, to simplify indexing for zfreeze; + // this requires subtracting 1 at the start. + LEA(32, remaining_reg, MDisp(ABI_PARAM3, -1)); MOV(64, R(base_reg), R(ABI_PARAM4)); @@ -407,9 +412,9 @@ void VertexLoaderX64::GenerateVertexLoader() MOV(32, MDisp(dst_reg, m_dst_ofs), R(scratch1)); // zfreeze - CMP(32, R(count_reg), Imm8(3)); - FixupBranch dont_store = J_CC(CC_A); - MOV(32, MPIC(VertexLoaderManager::position_matrix_index_cache.data(), count_reg, SCALE_4), + CMP(32, R(remaining_reg), Imm8(3)); + FixupBranch dont_store = J_CC(CC_AE); + MOV(32, MPIC(VertexLoaderManager::position_matrix_index_cache.data(), remaining_reg, SCALE_4), R(scratch1)); SetJumpTarget(dont_store); @@ -509,8 +514,8 @@ void VertexLoaderX64::GenerateVertexLoader() const u8* cont = GetCodePtr(); ADD(64, R(src_reg), Imm32(m_src_ofs)); - SUB(32, R(count_reg), Imm8(1)); - J_CC(CC_NZ, loop_start); + SUB(32, R(remaining_reg), Imm8(1)); + J_CC(CC_AE, loop_start); // Get the original count. POP(32, R(ABI_RETURN)); -- cgit v1.2.3 From 2a5c77f43ff1d69e78f12f13435a38c5eb2ed854 Mon Sep 17 00:00:00 2001 From: Pokechu22 Date: Wed, 13 Apr 2022 22:03:34 -0700 Subject: VideoCommon: Handle emboss texgen with only a single normal Fixes a large number of effects in Rogue Squadron 2 and 3. --- Source/Core/VideoCommon/VertexLoaderX64.cpp | 20 +++++++++++++++++++- 1 file changed, 19 insertions(+), 1 deletion(-) (limited to 'Source/Core/VideoCommon/VertexLoaderX64.cpp') diff --git a/Source/Core/VideoCommon/VertexLoaderX64.cpp b/Source/Core/VideoCommon/VertexLoaderX64.cpp index da52788d3c..aebba7680d 100644 --- a/Source/Core/VideoCommon/VertexLoaderX64.cpp +++ b/Source/Core/VideoCommon/VertexLoaderX64.cpp @@ -127,6 +127,22 @@ int VertexLoaderX64::ReadVertex(OpArg data, VertexComponentFormat attribute, Com MOVUPS(MPIC(VertexLoaderManager::position_cache.data(), scratch3, SCALE_4), coords); SetJumpTarget(dont_store); } + else if (native_format == &m_native_vtx_decl.normals[1]) + { + TEST(32, R(remaining_reg), R(remaining_reg)); + FixupBranch dont_store = J_CC(CC_NZ); + // For similar reasons, the cached tangent and binormal are 4 floats each + MOVUPS(MPIC(VertexLoaderManager::tangent_cache.data()), coords); + SetJumpTarget(dont_store); + } + else if (native_format == &m_native_vtx_decl.normals[2]) + { + CMP(32, R(remaining_reg), R(remaining_reg)); + FixupBranch dont_store = J_CC(CC_NZ); + // For similar reasons, the cached tangent and binormal are 4 floats each + MOVUPS(MPIC(VertexLoaderManager::binormal_cache.data()), coords); + SetJumpTarget(dont_store); + } }; int elem_size = GetElementSize(format); @@ -217,7 +233,9 @@ int VertexLoaderX64::ReadVertex(OpArg data, VertexComponentFormat attribute, Com dest.AddMemOffset(sizeof(float)); // zfreeze - if (native_format == &m_native_vtx_decl.position) + if (native_format == &m_native_vtx_decl.position || + native_format == &m_native_vtx_decl.normals[1] || + native_format == &m_native_vtx_decl.normals[2]) { if (cpu_info.bSSE4_1) { -- cgit v1.2.3