From 9e2f4dd7da306a011d3f34bad7c1808186a9bbd0 Mon Sep 17 00:00:00 2001 From: Tillmann Karras Date: Mon, 1 Jun 2015 19:43:35 +0200 Subject: VertexLoaderX64: revert 9da86092aeb1fda7470a661a36 I can't reproduce that it's actually faster and it will definitely be slower with position caching for zfreeze. --- Source/Core/VideoCommon/VertexLoaderX64.cpp | 42 ++++++++++++++++------------- 1 file changed, 23 insertions(+), 19 deletions(-) (limited to 'Source/Core/VideoCommon/VertexLoaderX64.cpp') diff --git a/Source/Core/VideoCommon/VertexLoaderX64.cpp b/Source/Core/VideoCommon/VertexLoaderX64.cpp index 9abde62b8b..ba29a7497a 100644 --- a/Source/Core/VideoCommon/VertexLoaderX64.cpp +++ b/Source/Core/VideoCommon/VertexLoaderX64.cpp @@ -77,7 +77,7 @@ OpArg VertexLoaderX64::GetVertexAddr(int array, u64 attribute) int VertexLoaderX64::ReadVertex(OpArg data, u64 attribute, int format, int count_in, int count_out, bool dequantize, u8 scaling_exponent, AttributeFormat* native_format) { - static const __m128i shuffle_lut[4][3] = { + static const __m128i shuffle_lut[5][3] = { {_mm_set_epi32(0xFFFFFFFFL, 0xFFFFFFFFL, 0xFFFFFFFFL, 0xFFFFFF00L), // 1x u8 _mm_set_epi32(0xFFFFFFFFL, 0xFFFFFFFFL, 0xFFFFFF01L, 0xFFFFFF00L), // 2x u8 _mm_set_epi32(0xFFFFFFFFL, 0xFFFFFF02L, 0xFFFFFF01L, 0xFFFFFF00L)}, // 3x u8 @@ -90,6 +90,9 @@ int VertexLoaderX64::ReadVertex(OpArg data, u64 attribute, int format, int count {_mm_set_epi32(0xFFFFFFFFL, 0xFFFFFFFFL, 0xFFFFFFFFL, 0x0001FFFFL), // 1x s16 _mm_set_epi32(0xFFFFFFFFL, 0xFFFFFFFFL, 0x0203FFFFL, 0x0001FFFFL), // 2x s16 _mm_set_epi32(0xFFFFFFFFL, 0x0405FFFFL, 0x0203FFFFL, 0x0001FFFFL)}, // 3x s16 + {_mm_set_epi32(0xFFFFFFFFL, 0xFFFFFFFFL, 0xFFFFFFFFL, 0x00010203L), // 1x float + _mm_set_epi32(0xFFFFFFFFL, 0xFFFFFFFFL, 0x04050607L, 0x00010203L), // 2x float + _mm_set_epi32(0xFFFFFFFFL, 0x08090A0BL, 0x04050607L, 0x00010203L)}, // 3x float }; static const __m128 scale_factors[32] = { _mm_set_ps1(1./(1u<< 0)), _mm_set_ps1(1./(1u<< 1)), _mm_set_ps1(1./(1u<< 2)), _mm_set_ps1(1./(1u<< 3)), @@ -119,21 +122,6 @@ int VertexLoaderX64::ReadVertex(OpArg data, u64 attribute, int format, int count if (attribute == DIRECT) m_src_ofs += load_bytes; - if (format == FORMAT_FLOAT) - { - // Floats don't need to be scaled or converted, - // so we can just load/swap/store them directly - // and return early. - for (int i = 0; i < count_in; i++) - { - LoadAndSwap(32, scratch3, data); - MOV(32, dest, R(scratch3)); - data.AddMemOffset(sizeof(float)); - dest.AddMemOffset(sizeof(float)); - } - return load_bytes; - } - if (cpu_info.bSSSE3) { if (load_bytes > 8) @@ -194,13 +182,29 @@ int VertexLoaderX64::ReadVertex(OpArg data, u64 attribute, int format, int count else PSRLD(coords, 16); break; + case FORMAT_FLOAT: + // Floats don't need to be scaled or converted, + // so we can just load/swap/store them directly + // and return early. + // (In SSSE3 we still need to store them.) + for (int i = 0; i < count_in; i++) + { + LoadAndSwap(32, scratch3, data); + MOV(32, dest, R(scratch3)); + data.AddMemOffset(sizeof(float)); + dest.AddMemOffset(sizeof(float)); + } + return load_bytes; } } - CVTDQ2PS(coords, R(coords)); + if (format != FORMAT_FLOAT) + { + CVTDQ2PS(coords, R(coords)); - if (dequantize && scaling_exponent) - MULPS(coords, MPIC(&scale_factors[scaling_exponent])); + if (dequantize && scaling_exponent) + MULPS(coords, MPIC(&scale_factors[scaling_exponent])); + } switch (count_out) { -- cgit v1.2.3 From 5ddd2cef6c5cc8e5414636d73365691de726b3d0 Mon Sep 17 00:00:00 2001 From: Tillmann Karras Date: Mon, 1 Jun 2015 19:58:27 +0200 Subject: zfreeze: cache vertex positions Suggested by degasus. --- Source/Core/VideoCommon/VertexLoaderX64.cpp | 47 +++++++++++++++++++++++++++++ 1 file changed, 47 insertions(+) (limited to 'Source/Core/VideoCommon/VertexLoaderX64.cpp') diff --git a/Source/Core/VideoCommon/VertexLoaderX64.cpp b/Source/Core/VideoCommon/VertexLoaderX64.cpp index ba29a7497a..a298d7e1dd 100644 --- a/Source/Core/VideoCommon/VertexLoaderX64.cpp +++ b/Source/Core/VideoCommon/VertexLoaderX64.cpp @@ -23,6 +23,11 @@ static const X64Reg base_reg = RBX; static const u8* memory_base_ptr = (u8*)&g_main_cp_state.array_strides; +static OpArg MPIC(const void* ptr, X64Reg scale_reg, int scale = SCALE_1) +{ + return MComplex(base_reg, scale_reg, scale, (s32)((u8*)ptr - memory_base_ptr)); +} + static OpArg MPIC(const void* ptr) { return MDisp(base_reg, (s32)((u8*)ptr - memory_base_ptr)); @@ -193,6 +198,31 @@ int VertexLoaderX64::ReadVertex(OpArg data, u64 attribute, int format, int count MOV(32, dest, R(scratch3)); data.AddMemOffset(sizeof(float)); dest.AddMemOffset(sizeof(float)); + + // zfreeze + if (native_format == &m_native_vtx_decl.position) + { + if (cpu_info.bSSE4_1) + { + PINSRD(coords, R(scratch3), i); + } + else + { + PINSRW(coords, R(scratch3), 2 * i + 0); + SHR(32, R(scratch3), Imm8(16)); + PINSRW(coords, R(scratch3), 2 * i + 1); + } + } + } + + // zfreeze + if (native_format == &m_native_vtx_decl.position) + { + CMP(32, R(count_reg), Imm8(3)); + FixupBranch dont_store = J_CC(CC_A); + LEA(32, scratch3, MScaled(count_reg, SCALE_4, -4)); + MOVUPS(MPIC(VertexLoaderManager::position_cache, scratch3, SCALE_4), coords); + SetJumpTarget(dont_store); } return load_bytes; } @@ -213,6 +243,16 @@ int VertexLoaderX64::ReadVertex(OpArg data, u64 attribute, int format, int count case 3: MOVUPS(dest, coords); break; } + // zfreeze + if (native_format == &m_native_vtx_decl.position) + { + CMP(32, R(count_reg), Imm8(3)); + FixupBranch dont_store = J_CC(CC_A); + LEA(32, scratch3, MScaled(count_reg, SCALE_4, -4)); + MOVUPS(MPIC(VertexLoaderManager::position_cache, scratch3, SCALE_4), coords); + SetJumpTarget(dont_store); + } + return load_bytes; } @@ -388,6 +428,13 @@ void VertexLoaderX64::GenerateVertexLoader() MOVZX(32, 8, scratch1, MDisp(src_reg, m_src_ofs)); AND(32, R(scratch1), Imm8(0x3F)); MOV(32, MDisp(dst_reg, m_dst_ofs), R(scratch1)); + + // zfreeze + CMP(32, R(count_reg), Imm8(3)); + FixupBranch dont_store = J_CC(CC_A); + MOV(32, MPIC(VertexLoaderManager::position_matrix_index - 1, count_reg, SCALE_4), R(scratch1)); + SetJumpTarget(dont_store); + m_native_components |= VB_HAS_POSMTXIDX; m_native_vtx_decl.posmtx.components = 4; m_native_vtx_decl.posmtx.enable = true; -- cgit v1.2.3