diff options
| author | Linktothepast <kostamarino@gmail.com> | 2015-09-28 16:50:35 +0300 |
|---|---|---|
| committer | Linktothepast <kostamarino@gmail.com> | 2015-09-28 16:50:35 +0300 |
| commit | 748be565b8df535ef558e74b38cd0f4d615f8a21 (patch) | |
| tree | b151efee0a87e99125f2401325a9bddecf84fc73 /Source/Core/VideoCommon | |
| parent | cf17fccdd3093a1b8476b10183b1c4bde36da74d (diff) | |
| parent | ce493b897d6d3735c930a8465cc0c26bbe5feb86 (diff) | |
Merge remote-tracking branch 'dolphin-emu/master' into gameinis
Diffstat (limited to 'Source/Core/VideoCommon')
60 files changed, 1168 insertions, 977 deletions
diff --git a/Source/Core/VideoCommon/AVIDump.cpp b/Source/Core/VideoCommon/AVIDump.cpp index fe5b0050e7..0424082709 100644 --- a/Source/Core/VideoCommon/AVIDump.cpp +++ b/Source/Core/VideoCommon/AVIDump.cpp @@ -480,12 +480,12 @@ void AVIDump::AddFrame(const u8* data, int width, int height) while (!error && got_packet) { // Write the compressed frame in the media file. - if (pkt.pts != AV_NOPTS_VALUE) + if (pkt.pts != (s64)AV_NOPTS_VALUE) { pkt.pts = av_rescale_q(pkt.pts, s_stream->codec->time_base, s_stream->time_base); } - if (pkt.dts != AV_NOPTS_VALUE) + if (pkt.dts != (s64)AV_NOPTS_VALUE) { pkt.dts = av_rescale_q(pkt.dts, s_stream->codec->time_base, s_stream->time_base); diff --git a/Source/Core/VideoCommon/BPFunctions.cpp b/Source/Core/VideoCommon/BPFunctions.cpp index 4150a7cf31..e4b5f54901 100644 --- a/Source/Core/VideoCommon/BPFunctions.cpp +++ b/Source/Core/VideoCommon/BPFunctions.cpp @@ -9,7 +9,6 @@ #include "VideoCommon/BPFunctions.h" #include "VideoCommon/RenderBase.h" -#include "VideoCommon/TextureCacheBase.h" #include "VideoCommon/VertexManagerBase.h" #include "VideoCommon/VertexShaderManager.h" #include "VideoCommon/VideoConfig.h" @@ -85,15 +84,6 @@ void SetColorMask() g_renderer->SetColorMask(); } -void CopyEFB(u32 dstAddr, const EFBRectangle& srcRect, - unsigned int dstFormat, PEControl::PixelFormat srcFormat, - bool isIntensity, bool scaleByHalf) -{ - // bpmem.zcontrol.pixel_format to PEControl::Z24 is when the game wants to copy from ZBuffer (Zbuffer uses 24-bit Format) - TextureCache::CopyRenderTargetToTexture(dstAddr, dstFormat, srcFormat, - srcRect, isIntensity, scaleByHalf); -} - /* Explanation of the magic behind ClearScreen: There's numerous possible formats for the pixel data in the EFB. However, in the HW accelerated backends we're always using RGBA8 diff --git a/Source/Core/VideoCommon/BPFunctions.h b/Source/Core/VideoCommon/BPFunctions.h index 890f734deb..126d9f1d7b 100644 --- a/Source/Core/VideoCommon/BPFunctions.h +++ b/Source/Core/VideoCommon/BPFunctions.h @@ -23,9 +23,6 @@ void SetBlendMode(); void SetDitherMode(); void SetLogicOpMode(); void SetColorMask(); -void CopyEFB(u32 dstAddr, const EFBRectangle& srcRect, - unsigned int dstFormat, PEControl::PixelFormat srcFormat, - bool isIntensity, bool scaleByHalf); void ClearScreen(const EFBRectangle &rc); void OnPixelFormatChange(); void SetInterlacingMode(const BPCmd &bp); diff --git a/Source/Core/VideoCommon/BPMemory.h b/Source/Core/VideoCommon/BPMemory.h index 50a95791ce..14ac732edf 100644 --- a/Source/Core/VideoCommon/BPMemory.h +++ b/Source/Core/VideoCommon/BPMemory.h @@ -11,154 +11,239 @@ #pragma pack(4) -#define BPMEM_GENMODE 0x00 -#define BPMEM_DISPLAYCOPYFILTER 0x01 // 0x01 + 4 -#define BPMEM_IND_MTXA 0x06 // 0x06 + (3 * 3) -#define BPMEM_IND_MTXB 0x07 // 0x07 + (3 * 3) -#define BPMEM_IND_MTXC 0x08 // 0x08 + (3 * 3) -#define BPMEM_IND_IMASK 0x0F -#define BPMEM_IND_CMD 0x10 // 0x10 + 16 -#define BPMEM_SCISSORTL 0x20 -#define BPMEM_SCISSORBR 0x21 -#define BPMEM_LINEPTWIDTH 0x22 -#define BPMEM_PERF0_TRI 0x23 -#define BPMEM_PERF0_QUAD 0x24 -#define BPMEM_RAS1_SS0 0x25 -#define BPMEM_RAS1_SS1 0x26 -#define BPMEM_IREF 0x27 -#define BPMEM_TREF 0x28 // 0x28 + 8 -#define BPMEM_SU_SSIZE 0x30 // 0x30 + (2 * 8) -#define BPMEM_SU_TSIZE 0x31 // 0x31 + (2 * 8) -#define BPMEM_ZMODE 0x40 -#define BPMEM_BLENDMODE 0x41 -#define BPMEM_CONSTANTALPHA 0x42 -#define BPMEM_ZCOMPARE 0x43 -#define BPMEM_FIELDMASK 0x44 -#define BPMEM_SETDRAWDONE 0x45 -#define BPMEM_BUSCLOCK0 0x46 -#define BPMEM_PE_TOKEN_ID 0x47 -#define BPMEM_PE_TOKEN_INT_ID 0x48 -#define BPMEM_EFB_TL 0x49 -#define BPMEM_EFB_BR 0x4A -#define BPMEM_EFB_ADDR 0x4B -#define BPMEM_MIPMAP_STRIDE 0x4D -#define BPMEM_COPYYSCALE 0x4E -#define BPMEM_CLEAR_AR 0x4F -#define BPMEM_CLEAR_GB 0x50 -#define BPMEM_CLEAR_Z 0x51 -#define BPMEM_TRIGGER_EFB_COPY 0x52 -#define BPMEM_COPYFILTER0 0x53 -#define BPMEM_COPYFILTER1 0x54 -#define BPMEM_CLEARBBOX1 0x55 -#define BPMEM_CLEARBBOX2 0x56 -#define BPMEM_CLEAR_PIXEL_PERF 0x57 -#define BPMEM_REVBITS 0x58 -#define BPMEM_SCISSOROFFSET 0x59 -#define BPMEM_PRELOAD_ADDR 0x60 -#define BPMEM_PRELOAD_TMEMEVEN 0x61 -#define BPMEM_PRELOAD_TMEMODD 0x62 -#define BPMEM_PRELOAD_MODE 0x63 -#define BPMEM_LOADTLUT0 0x64 -#define BPMEM_LOADTLUT1 0x65 -#define BPMEM_TEXINVALIDATE 0x66 -#define BPMEM_PERF1 0x67 -#define BPMEM_FIELDMODE 0x68 -#define BPMEM_BUSCLOCK1 0x69 -#define BPMEM_TX_SETMODE0 0x80 // 0x80 + 4 -#define BPMEM_TX_SETMODE1 0x84 // 0x84 + 4 -#define BPMEM_TX_SETIMAGE0 0x88 // 0x88 + 4 -#define BPMEM_TX_SETIMAGE1 0x8C // 0x8C + 4 -#define BPMEM_TX_SETIMAGE2 0x90 // 0x90 + 4 -#define BPMEM_TX_SETIMAGE3 0x94 // 0x94 + 4 -#define BPMEM_TX_SETTLUT 0x98 // 0x98 + 4 -#define BPMEM_TX_SETMODE0_4 0xA0 // 0xA0 + 4 -#define BPMEM_TX_SETMODE1_4 0xA4 // 0xA4 + 4 -#define BPMEM_TX_SETIMAGE0_4 0xA8 // 0xA8 + 4 -#define BPMEM_TX_SETIMAGE1_4 0xAC // 0xA4 + 4 -#define BPMEM_TX_SETIMAGE2_4 0xB0 // 0xB0 + 4 -#define BPMEM_TX_SETIMAGE3_4 0xB4 // 0xB4 + 4 -#define BPMEM_TX_SETTLUT_4 0xB8 // 0xB8 + 4 -#define BPMEM_TEV_COLOR_ENV 0xC0 // 0xC0 + (2 * 16) -#define BPMEM_TEV_ALPHA_ENV 0xC1 // 0xC1 + (2 * 16) -#define BPMEM_TEV_COLOR_RA 0xE0 // 0xE0 + (2 * 4) -#define BPMEM_TEV_COLOR_BG 0xE1 // 0xE1 + (2 * 4) -#define BPMEM_FOGRANGE 0xE8 // 0xE8 + 6 -#define BPMEM_FOGPARAM0 0xEE -#define BPMEM_FOGBMAGNITUDE 0xEF -#define BPMEM_FOGBEXPONENT 0xF0 -#define BPMEM_FOGPARAM3 0xF1 -#define BPMEM_FOGCOLOR 0xF2 -#define BPMEM_ALPHACOMPARE 0xF3 -#define BPMEM_BIAS 0xF4 -#define BPMEM_ZTEX2 0xF5 -#define BPMEM_TEV_KSEL 0xF6 // 0xF6 + 8 -#define BPMEM_BP_MASK 0xFE +enum +{ + BPMEM_GENMODE = 0x00, + BPMEM_DISPLAYCOPYFILTER = 0x01, // 0x01 + 4 + BPMEM_IND_MTXA = 0x06, // 0x06 + (3 * 3) + BPMEM_IND_MTXB = 0x07, // 0x07 + (3 * 3) + BPMEM_IND_MTXC = 0x08, // 0x08 + (3 * 3) + BPMEM_IND_IMASK = 0x0F, + BPMEM_IND_CMD = 0x10, // 0x10 + 16 + BPMEM_SCISSORTL = 0x20, + BPMEM_SCISSORBR = 0x21, + BPMEM_LINEPTWIDTH = 0x22, + BPMEM_PERF0_TRI = 0x23, + BPMEM_PERF0_QUAD = 0x24, + BPMEM_RAS1_SS0 = 0x25, + BPMEM_RAS1_SS1 = 0x26, + BPMEM_IREF = 0x27, + BPMEM_TREF = 0x28, // 0x28 + 8 + BPMEM_SU_SSIZE = 0x30, // 0x30 + (2 * 8) + BPMEM_SU_TSIZE = 0x31, // 0x31 + (2 * 8) + BPMEM_ZMODE = 0x40, + BPMEM_BLENDMODE = 0x41, + BPMEM_CONSTANTALPHA = 0x42, + BPMEM_ZCOMPARE = 0x43, + BPMEM_FIELDMASK = 0x44, + BPMEM_SETDRAWDONE = 0x45, + BPMEM_BUSCLOCK0 = 0x46, + BPMEM_PE_TOKEN_ID = 0x47, + BPMEM_PE_TOKEN_INT_ID = 0x48, + BPMEM_EFB_TL = 0x49, + BPMEM_EFB_BR = 0x4A, + BPMEM_EFB_ADDR = 0x4B, + BPMEM_MIPMAP_STRIDE = 0x4D, + BPMEM_COPYYSCALE = 0x4E, + BPMEM_CLEAR_AR = 0x4F, + BPMEM_CLEAR_GB = 0x50, + BPMEM_CLEAR_Z = 0x51, + BPMEM_TRIGGER_EFB_COPY = 0x52, + BPMEM_COPYFILTER0 = 0x53, + BPMEM_COPYFILTER1 = 0x54, + BPMEM_CLEARBBOX1 = 0x55, + BPMEM_CLEARBBOX2 = 0x56, + BPMEM_CLEAR_PIXEL_PERF = 0x57, + BPMEM_REVBITS = 0x58, + BPMEM_SCISSOROFFSET = 0x59, + BPMEM_PRELOAD_ADDR = 0x60, + BPMEM_PRELOAD_TMEMEVEN = 0x61, + BPMEM_PRELOAD_TMEMODD = 0x62, + BPMEM_PRELOAD_MODE = 0x63, + BPMEM_LOADTLUT0 = 0x64, + BPMEM_LOADTLUT1 = 0x65, + BPMEM_TEXINVALIDATE = 0x66, + BPMEM_PERF1 = 0x67, + BPMEM_FIELDMODE = 0x68, + BPMEM_BUSCLOCK1 = 0x69, + BPMEM_TX_SETMODE0 = 0x80, // 0x80 + 4 + BPMEM_TX_SETMODE1 = 0x84, // 0x84 + 4 + BPMEM_TX_SETIMAGE0 = 0x88, // 0x88 + 4 + BPMEM_TX_SETIMAGE1 = 0x8C, // 0x8C + 4 + BPMEM_TX_SETIMAGE2 = 0x90, // 0x90 + 4 + BPMEM_TX_SETIMAGE3 = 0x94, // 0x94 + 4 + BPMEM_TX_SETTLUT = 0x98, // 0x98 + 4 + BPMEM_TX_SETMODE0_4 = 0xA0, // 0xA0 + 4 + BPMEM_TX_SETMODE1_4 = 0xA4, // 0xA4 + 4 + BPMEM_TX_SETIMAGE0_4 = 0xA8, // 0xA8 + 4 + BPMEM_TX_SETIMAGE1_4 = 0xAC, // 0xA4 + 4 + BPMEM_TX_SETIMAGE2_4 = 0xB0, // 0xB0 + 4 + BPMEM_TX_SETIMAGE3_4 = 0xB4, // 0xB4 + 4 + BPMEM_TX_SETTLUT_4 = 0xB8, // 0xB8 + 4 + BPMEM_TEV_COLOR_ENV = 0xC0, // 0xC0 + (2 * 16) + BPMEM_TEV_ALPHA_ENV = 0xC1, // 0xC1 + (2 * 16) + BPMEM_TEV_COLOR_RA = 0xE0, // 0xE0 + (2 * 4) + BPMEM_TEV_COLOR_BG = 0xE1, // 0xE1 + (2 * 4) + BPMEM_FOGRANGE = 0xE8, // 0xE8 + 6 + BPMEM_FOGPARAM0 = 0xEE, + BPMEM_FOGBMAGNITUDE = 0xEF, + BPMEM_FOGBEXPONENT = 0xF0, + BPMEM_FOGPARAM3 = 0xF1, + BPMEM_FOGCOLOR = 0xF2, + BPMEM_ALPHACOMPARE = 0xF3, + BPMEM_BIAS = 0xF4, + BPMEM_ZTEX2 = 0xF5, + BPMEM_TEV_KSEL = 0xF6, // 0xF6 + 8 + BPMEM_BP_MASK = 0xFE, +}; // Tev/combiner things -#define TEVSCALE_1 0 -#define TEVSCALE_2 1 -#define TEVSCALE_4 2 -#define TEVDIVIDE_2 3 - -#define TEVCMP_R8 0 -#define TEVCMP_GR16 1 -#define TEVCMP_BGR24 2 -#define TEVCMP_RGB8 3 - -#define TEVOP_ADD 0 -#define TEVOP_SUB 1 -#define TEVCMP_R8_GT 8 -#define TEVCMP_R8_EQ 9 -#define TEVCMP_GR16_GT 10 -#define TEVCMP_GR16_EQ 11 -#define TEVCMP_BGR24_GT 12 -#define TEVCMP_BGR24_EQ 13 -#define TEVCMP_RGB8_GT 14 -#define TEVCMP_RGB8_EQ 15 -#define TEVCMP_A8_GT 14 -#define TEVCMP_A8_EQ 15 - -#define TEVCOLORARG_CPREV 0 -#define TEVCOLORARG_APREV 1 -#define TEVCOLORARG_C0 2 -#define TEVCOLORARG_A0 3 -#define TEVCOLORARG_C1 4 -#define TEVCOLORARG_A1 5 -#define TEVCOLORARG_C2 6 -#define TEVCOLORARG_A2 7 -#define TEVCOLORARG_TEXC 8 -#define TEVCOLORARG_TEXA 9 -#define TEVCOLORARG_RASC 10 -#define TEVCOLORARG_RASA 11 -#define TEVCOLORARG_ONE 12 -#define TEVCOLORARG_HALF 13 -#define TEVCOLORARG_KONST 14 -#define TEVCOLORARG_ZERO 15 - -#define TEVALPHAARG_APREV 0 -#define TEVALPHAARG_A0 1 -#define TEVALPHAARG_A1 2 -#define TEVALPHAARG_A2 3 -#define TEVALPHAARG_TEXA 4 -#define TEVALPHAARG_RASA 5 -#define TEVALPHAARG_KONST 6 -#define TEVALPHAARG_ZERO 7 - -#define GX_TEVPREV 0 -#define GX_TEVREG0 1 -#define GX_TEVREG1 2 -#define GX_TEVREG2 3 - -#define ZTEXTURE_DISABLE 0 -#define ZTEXTURE_ADD 1 -#define ZTEXTURE_REPLACE 2 - -#define TevBias_ZERO 0 -#define TevBias_ADDHALF 1 -#define TevBias_SUBHALF 2 -#define TevBias_COMPARE 3 +// TEV scaling type +enum : u32 +{ + TEVSCALE_1 = 0, + TEVSCALE_2 = 1, + TEVSCALE_4 = 2, + TEVDIVIDE_2 = 3 +}; + +enum : u32 +{ + TEVCMP_R8 = 0, + TEVCMP_GR16 = 1, + TEVCMP_BGR24 = 2, + TEVCMP_RGB8 = 3 +}; + +// TEV combiner operator +enum : u32 +{ + TEVOP_ADD = 0, + TEVOP_SUB = 1, + TEVCMP_R8_GT = 8, + TEVCMP_R8_EQ = 9, + TEVCMP_GR16_GT = 10, + TEVCMP_GR16_EQ = 11, + TEVCMP_BGR24_GT = 12, + TEVCMP_BGR24_EQ = 13, + TEVCMP_RGB8_GT = 14, + TEVCMP_RGB8_EQ = 15, + TEVCMP_A8_GT = TEVCMP_RGB8_GT, + TEVCMP_A8_EQ = TEVCMP_RGB8_EQ +}; + +// TEV color combiner input +enum : u32 +{ + TEVCOLORARG_CPREV = 0, + TEVCOLORARG_APREV = 1, + TEVCOLORARG_C0 = 2, + TEVCOLORARG_A0 = 3, + TEVCOLORARG_C1 = 4, + TEVCOLORARG_A1 = 5, + TEVCOLORARG_C2 = 6, + TEVCOLORARG_A2 = 7, + TEVCOLORARG_TEXC = 8, + TEVCOLORARG_TEXA = 9, + TEVCOLORARG_RASC = 10, + TEVCOLORARG_RASA = 11, + TEVCOLORARG_ONE = 12, + TEVCOLORARG_HALF = 13, + TEVCOLORARG_KONST = 14, + TEVCOLORARG_ZERO = 15 +}; + +// TEV alpha combiner input +enum : u32 +{ + TEVALPHAARG_APREV = 0, + TEVALPHAARG_A0 = 1, + TEVALPHAARG_A1 = 2, + TEVALPHAARG_A2 = 3, + TEVALPHAARG_TEXA = 4, + TEVALPHAARG_RASA = 5, + TEVALPHAARG_KONST = 6, + TEVALPHAARG_ZERO = 7 +}; + +// TEV output registers +enum : u32 +{ + GX_TEVPREV = 0, + GX_TEVREG0 = 1, + GX_TEVREG1 = 2, + GX_TEVREG2 = 3 +}; + +// Z-texture formats +enum : u32 +{ + TEV_ZTEX_TYPE_U8 = 0, + TEV_ZTEX_TYPE_U16 = 1, + TEV_ZTEX_TYPE_U24 = 2 +}; + +// Z texture operator +enum : u32 +{ + ZTEXTURE_DISABLE = 0, + ZTEXTURE_ADD = 1, + ZTEXTURE_REPLACE = 2 +}; + +// TEV bias value +enum : u32 +{ + TEVBIAS_ZERO = 0, + TEVBIAS_ADDHALF = 1, + TEVBIAS_SUBHALF = 2, + TEVBIAS_COMPARE = 3 +}; + +// Indirect texture format +enum : u32 +{ + ITF_8 = 0, + ITF_5 = 1, + ITF_4 = 2, + ITF_3 = 3 +}; + +// Indirect texture bias +enum : u32 +{ + ITB_NONE = 0, + ITB_S = 1, + ITB_T = 2, + ITB_ST = 3, + ITB_U = 4, + ITB_SU = 5, + ITB_TU = 6, + ITB_STU = 7 +}; + +// Indirect texture bump alpha +enum : u32 +{ + ITBA_OFF = 0, + ITBA_S = 1, + ITBA_T = 2, + ITBA_U = 3 +}; + +// Indirect texture wrap value +enum : u32 +{ + ITW_OFF = 0, + ITW_256 = 1, + ITW_128 = 2, + ITW_64 = 3, + ITW_32 = 4, + ITW_16 = 5, + ITW_0 = 6 +}; union IND_MTXA { @@ -213,32 +298,6 @@ union IND_IMASK u32 hex; }; -#define TEVSELCC_CPREV 0 -#define TEVSELCC_APREV 1 -#define TEVSELCC_C0 2 -#define TEVSELCC_A0 3 -#define TEVSELCC_C1 4 -#define TEVSELCC_A1 5 -#define TEVSELCC_C2 6 -#define TEVSELCC_A2 7 -#define TEVSELCC_TEXC 8 -#define TEVSELCC_TEXA 9 -#define TEVSELCC_RASC 10 -#define TEVSELCC_RASA 11 -#define TEVSELCC_ONE 12 -#define TEVSELCC_HALF 13 -#define TEVSELCC_KONST 14 -#define TEVSELCC_ZERO 15 - -#define TEVSELCA_APREV 0 -#define TEVSELCA_A0 1 -#define TEVSELCA_A1 2 -#define TEVSELCA_A2 3 -#define TEVSELCA_TEXA 4 -#define TEVSELCA_RASA 5 -#define TEVSELCA_KONST 6 -#define TEVSELCA_ZERO 7 - struct TevStageCombiner { union ColorCombiner @@ -285,33 +344,6 @@ struct TevStageCombiner AlphaCombiner alphaC; }; -#define ITF_8 0 -#define ITF_5 1 -#define ITF_4 2 -#define ITF_3 3 - -#define ITB_NONE 0 -#define ITB_S 1 -#define ITB_T 2 -#define ITB_ST 3 -#define ITB_U 4 -#define ITB_SU 5 -#define ITB_TU 6 -#define ITB_STU 7 - -#define ITBA_OFF 0 -#define ITBA_S 1 -#define ITBA_T 2 -#define ITBA_U 3 - -#define ITW_OFF 0 -#define ITW_256 1 -#define ITW_128 2 -#define ITW_64 3 -#define ITW_32 4 -#define ITW_16 5 -#define ITW_0 6 - // several discoveries: // GXSetTevIndBumpST(tevstage, indstage, matrixind) // if ( matrix == 2 ) realmat = 6; // 10 @@ -513,16 +545,6 @@ union ZTex2 u32 hex; }; -// Z-texture types (formats) -#define TEV_ZTEX_TYPE_U8 0 -#define TEV_ZTEX_TYPE_U16 1 -#define TEV_ZTEX_TYPE_U24 2 - -#define TEV_ZTEX_DISABLE 0 -#define TEV_ZTEX_ADD 1 -#define TEV_ZTEX_REPLACE 2 - - struct FourTexUnits { TexMode0 texMode0[4]; diff --git a/Source/Core/VideoCommon/BPStructs.cpp b/Source/Core/VideoCommon/BPStructs.cpp index 410d4d44cc..4e3df7cc00 100644 --- a/Source/Core/VideoCommon/BPStructs.cpp +++ b/Source/Core/VideoCommon/BPStructs.cpp @@ -8,6 +8,7 @@ #include "Common/Thread.h" #include "Core/ConfigManager.h" #include "Core/Core.h" +#include "Core/FifoPlayer/FifoRecorder.h" #include "Core/HW/Memmap.h" #include "VideoCommon/BoundingBox.h" @@ -20,6 +21,7 @@ #include "VideoCommon/PixelShaderManager.h" #include "VideoCommon/RenderBase.h" #include "VideoCommon/Statistics.h" +#include "VideoCommon/TextureCacheBase.h" #include "VideoCommon/TextureDecoder.h" #include "VideoCommon/VertexShaderManager.h" #include "VideoCommon/VideoCommon.h" @@ -205,6 +207,7 @@ static void BPWritten(const BPCmd& bp) // The values in bpmem.copyTexSrcXY and bpmem.copyTexSrcWH are updated in case 0x49 and 0x4a in this function u32 destAddr = bpmem.copyTexDest << 5; + u32 destStride = bpmem.copyMipMapStrideChannels << 5; EFBRectangle srcRect; srcRect.left = (int)bpmem.copyTexSrcXY.x; @@ -223,8 +226,9 @@ static void BPWritten(const BPCmd& bp) if (g_ActiveConfig.bShowEFBCopyRegions) stats.efb_regions.push_back(srcRect); - CopyEFB(destAddr, srcRect, - PE_copy.tp_realFormat(), bpmem.zcontrol.pixel_format, + // bpmem.zcontrol.pixel_format to PEControl::Z24 is when the game wants to copy from ZBuffer (Zbuffer uses 24-bit Format) + TextureCache::CopyRenderTargetToTexture(destAddr, PE_copy.tp_realFormat(), destStride, + bpmem.zcontrol.pixel_format, srcRect, !!PE_copy.intensity_fmt, !!PE_copy.half_scale); } else @@ -241,7 +245,7 @@ static void BPWritten(const BPCmd& bp) else yScale = (float)bpmem.dispcopyyscale / 256.0f; - float num_xfb_lines = ((bpmem.copyTexSrcWH.y + 1.0f) * yScale); + float num_xfb_lines = 1.0f + bpmem.copyTexSrcWH.y * yScale; u32 height = static_cast<u32>(num_xfb_lines); if (height > MAX_XFB_HEIGHT) @@ -251,9 +255,9 @@ static void BPWritten(const BPCmd& bp) height = MAX_XFB_HEIGHT; } - u32 width = bpmem.copyMipMapStrideChannels << 4; - - Renderer::RenderToXFB(destAddr, srcRect, width, height, s_gammaLUT[PE_copy.gamma]); + DEBUG_LOG(VIDEO, "RenderToXFB: destAddr: %08x | srcRect {%d %d %d %d} | fbWidth: %u | fbStride: %u | fbHeight: %u", + destAddr, srcRect.left, srcRect.top, srcRect.right, srcRect.bottom, bpmem.copyTexSrcWH.x + 1, destStride, height); + Renderer::RenderToXFB(destAddr, srcRect, destStride, height, s_gammaLUT[PE_copy.gamma]); } // Clear the rectangular region after copying it. @@ -278,6 +282,9 @@ static void BPWritten(const BPCmd& bp) Memory::CopyFromEmu(texMem + tlutTMemAddr, addr, tlutXferCount); + if (g_bRecordFifoData) + FifoRecorder::GetInstance().UseMemory(addr, tlutXferCount, MemoryUpdate::TMEM); + return; } case BPMEM_FOGRANGE: // Fog Settings Control @@ -452,15 +459,16 @@ static void BPWritten(const BPCmd& bp) BPS_TmemConfig& tmem_cfg = bpmem.tmem_config; u32 src_addr = tmem_cfg.preload_addr << 5; // TODO: Should we add mask here on GC? - u32 size = tmem_cfg.preload_tile_info.count * TMEM_LINE_SIZE; + u32 bytes_read = 0; u32 tmem_addr_even = tmem_cfg.preload_tmem_even * TMEM_LINE_SIZE; if (tmem_cfg.preload_tile_info.type != 3) { - if (tmem_addr_even + size > TMEM_SIZE) - size = TMEM_SIZE - tmem_addr_even; + bytes_read = tmem_cfg.preload_tile_info.count * TMEM_LINE_SIZE; + if (tmem_addr_even + bytes_read > TMEM_SIZE) + bytes_read = TMEM_SIZE - tmem_addr_even; - Memory::CopyFromEmu(texMem + tmem_addr_even, src_addr, size); + Memory::CopyFromEmu(texMem + tmem_addr_even, src_addr, bytes_read); } else // RGBA8 tiles (and CI14, but that might just be stupid libogc!) { @@ -471,18 +479,19 @@ static void BPWritten(const BPCmd& bp) for (u32 i = 0; i < tmem_cfg.preload_tile_info.count; ++i) { - if (tmem_addr_even + TMEM_LINE_SIZE > TMEM_SIZE || - tmem_addr_odd + TMEM_LINE_SIZE > TMEM_SIZE) - return; + if (tmem_addr_even + TMEM_LINE_SIZE > TMEM_SIZE || tmem_addr_odd + TMEM_LINE_SIZE > TMEM_SIZE) + break; - // TODO: This isn't very optimised, does a whole lot of small memcpys - memcpy(texMem + tmem_addr_even, src_ptr, TMEM_LINE_SIZE); - memcpy(texMem + tmem_addr_odd, src_ptr + TMEM_LINE_SIZE, TMEM_LINE_SIZE); + memcpy(texMem + tmem_addr_even, src_ptr + bytes_read, TMEM_LINE_SIZE); + memcpy(texMem + tmem_addr_odd, src_ptr + bytes_read + TMEM_LINE_SIZE, TMEM_LINE_SIZE); tmem_addr_even += TMEM_LINE_SIZE; tmem_addr_odd += TMEM_LINE_SIZE; - src_ptr += TMEM_LINE_SIZE * 2; + bytes_read += TMEM_LINE_SIZE * 2; } } + + if (g_bRecordFifoData) + FifoRecorder::GetInstance().UseMemory(src_addr, bytes_read, MemoryUpdate::TMEM); } return; diff --git a/Source/Core/VideoCommon/BoundingBox.cpp b/Source/Core/VideoCommon/BoundingBox.cpp index f1bf5d066e..9336a6a671 100644 --- a/Source/Core/VideoCommon/BoundingBox.cpp +++ b/Source/Core/VideoCommon/BoundingBox.cpp @@ -2,15 +2,9 @@ // Licensed under GPLv2+ // Refer to the license.txt file included. - - -#include "VideoBackends/Software/Clipper.h" -#include "VideoBackends/Software/Rasterizer.h" -#include "VideoBackends/Software/SetupUnit.h" -#include "VideoBackends/Software/TransformUnit.h" +#include "Common/ChunkFile.h" +#include "Common/CommonTypes.h" #include "VideoCommon/BoundingBox.h" -#include "VideoCommon/PixelShaderManager.h" - namespace BoundingBox { diff --git a/Source/Core/VideoCommon/CommandProcessor.h b/Source/Core/VideoCommon/CommandProcessor.h index d5061efd02..2618a19801 100644 --- a/Source/Core/VideoCommon/CommandProcessor.h +++ b/Source/Core/VideoCommon/CommandProcessor.h @@ -11,8 +11,6 @@ class PointerWrap; namespace MMIO { class Mapping; } -extern bool MT; - namespace CommandProcessor { diff --git a/Source/Core/VideoCommon/DataReader.h b/Source/Core/VideoCommon/DataReader.h index 0737a9bf1d..bb4914db8c 100644 --- a/Source/Core/VideoCommon/DataReader.h +++ b/Source/Core/VideoCommon/DataReader.h @@ -4,7 +4,9 @@ #pragma once -#include "Common/Common.h" +#include <cstring> +#include "Common/CommonFuncs.h" +#include "Common/CommonTypes.h" class DataReader { @@ -33,9 +35,12 @@ public: template <typename T, bool swapped = true> __forceinline T Peek(int offset = 0) { - T data = *(T*)(buffer + offset); + T data; + std::memcpy(&data, &buffer[offset], sizeof(T)); + if (swapped) data = Common::FromBigEndian(data); + return data; } @@ -50,7 +55,8 @@ public: { if (swapped) data = Common::FromBigEndian(data); - *(T*)(buffer) = data; + + std::memcpy(buffer, &data, sizeof(T)); buffer += sizeof(T); } diff --git a/Source/Core/VideoCommon/Debugger.cpp b/Source/Core/VideoCommon/Debugger.cpp index 0c88be57f5..24ec679400 100644 --- a/Source/Core/VideoCommon/Debugger.cpp +++ b/Source/Core/VideoCommon/Debugger.cpp @@ -6,6 +6,7 @@ #include "Common/FileUtil.h" #include "Common/IniFile.h" +#include "Common/Thread.h" #include "VideoCommon/Debugger.h" #include "VideoCommon/NativeVertexFormat.h" @@ -56,7 +57,7 @@ void GFXDebuggerCheckAndPause(bool update) while ( GFXDebuggerPauseFlag ) { if (update) GFXDebuggerUpdateScreen(); - SLEEP(5); + Common::SleepCurrentThread(5); } g_pdebugger->OnContinue(); } @@ -81,7 +82,7 @@ void GFXDebuggerBase::DumpPixelShader(const std::string& path) const std::string filename = StringFromFormat("%sdump_ps.txt", path.c_str()); std::string output; - bool useDstAlpha = !g_ActiveConfig.bDstAlphaPass && bpmem.dstalpha.enable && bpmem.blendmode.alphaupdate && bpmem.zcontrol.pixel_format == PEControl::RGBA6_Z24; + bool useDstAlpha = bpmem.dstalpha.enable && bpmem.blendmode.alphaupdate && bpmem.zcontrol.pixel_format == PEControl::RGBA6_Z24; if (!useDstAlpha) { output = "Destination alpha disabled:\n"; diff --git a/Source/Core/VideoCommon/DriverDetails.cpp b/Source/Core/VideoCommon/DriverDetails.cpp index d3b1883cb3..cb9d01ba35 100644 --- a/Source/Core/VideoCommon/DriverDetails.cpp +++ b/Source/Core/VideoCommon/DriverDetails.cpp @@ -42,20 +42,16 @@ namespace DriverDetails // This is a list of all known bugs for each vendor // We use this to check if the device and driver has a issue static BugInfo m_known_bugs[] = { - {OS_ALL, VENDOR_QUALCOMM, DRIVER_QUALCOMM, -1, BUG_NODYNUBOACCESS, 14.0, 94.0, true}, - {OS_ALL, VENDOR_QUALCOMM, DRIVER_QUALCOMM, -1, BUG_BROKENCENTROID, 14.0, 46.0, true}, - {OS_ALL, VENDOR_QUALCOMM, DRIVER_QUALCOMM, -1, BUG_BROKENINFOLOG, -1.0, 46.0, true}, - {OS_ALL, VENDOR_QUALCOMM, DRIVER_QUALCOMM, -1, BUG_ANNIHILATEDUBOS, 41.0, 46.0, true}, - {OS_ALL, VENDOR_QUALCOMM, DRIVER_QUALCOMM, -1, BUG_BROKENSWAP, -1.0, 46.0, true}, {OS_ALL, VENDOR_QUALCOMM, DRIVER_QUALCOMM, -1, BUG_BROKENBUFFERSTREAM, -1.0, -1.0, true}, - {OS_ALL, VENDOR_QUALCOMM, DRIVER_QUALCOMM, -1, BUG_BROKENTEXTURESIZE, -1.0, 65.0, true}, - {OS_ALL, VENDOR_QUALCOMM, DRIVER_QUALCOMM, -1, BUG_BROKENATTRIBUTELESS, -1.0, 94.0, true}, {OS_ALL, VENDOR_QUALCOMM, DRIVER_QUALCOMM, -1, BUG_BROKENNEGATEDBOOLEAN,-1.0, -1.0, true}, - {OS_ALL, VENDOR_QUALCOMM, DRIVER_QUALCOMM, -1, BUG_BROKENIVECSHIFTS, -1.0, 46.0, true}, + {OS_ALL, VENDOR_QUALCOMM, DRIVER_QUALCOMM, -1, BUG_BROKENGLES31, -1.0, -1.0, true}, {OS_ALL, VENDOR_ARM, DRIVER_ARM, -1, BUG_BROKENBUFFERSTREAM, -1.0, -1.0, true}, + {OS_ALL, VENDOR_ARM, DRIVER_ARM, -1, BUG_BROKENVSYNC, -1.0, -1.0, true}, + {OS_ALL, VENDOR_IMGTEC, DRIVER_IMGTEC, -1, BUG_BROKENBUFFERSTREAM, -1.0, -1.0, true}, {OS_ALL, VENDOR_MESA, DRIVER_NOUVEAU, -1, BUG_BROKENUBO, 900, 916, true}, {OS_ALL, VENDOR_MESA, DRIVER_R600, -1, BUG_BROKENUBO, 900, 913, true}, {OS_ALL, VENDOR_MESA, DRIVER_I965, -1, BUG_BROKENUBO, 900, 920, true}, + {OS_ALL, VENDOR_MESA, DRIVER_ALL, -1, BUG_BROKENCOPYIMAGE, -1.0, 1064.0, true}, {OS_LINUX, VENDOR_ATI, DRIVER_ATI, -1, BUG_BROKENPINNEDMEMORY, -1.0, -1.0, true}, {OS_LINUX, VENDOR_NVIDIA, DRIVER_NVIDIA, -1, BUG_BROKENBUFFERSTORAGE, -1.0, 33138.0, true}, {OS_OSX, VENDOR_INTEL, DRIVER_INTEL, 3000, BUG_PRIMITIVERESTART, -1.0, -1.0, true}, @@ -105,7 +101,7 @@ namespace DriverDetails ( bug.m_versionstart <= m_version || bug.m_versionstart == -1 ) && ( bug.m_versionend > m_version || bug.m_versionend == -1 ) ) - m_bugs.insert(std::make_pair(bug.m_bug, bug)); + m_bugs.emplace(bug.m_bug, bug); } } diff --git a/Source/Core/VideoCommon/DriverDetails.h b/Source/Core/VideoCommon/DriverDetails.h index cdc5026dc2..1d49717499 100644 --- a/Source/Core/VideoCommon/DriverDetails.h +++ b/Source/Core/VideoCommon/DriverDetails.h @@ -58,32 +58,6 @@ namespace DriverDetails // This'll ensure we know exactly what the issue is. enum Bug { - // Bug: No Dynamic UBO array object access - // Affected Devices: Qualcomm/Adreno - // Started Version: 14 - // Ended Version: 95 - // Accessing UBO array members dynamically causes the Adreno shader compiler to crash - // Errors out with "Internal Error" - // With v53 video drivers, dynamic member access "works." It works to the extent that it doesn't crash. - // With v95 drivers everything works as it should. - BUG_NODYNUBOACCESS = 0, - // Bug: Centroid is broken in shaders - // Affected devices: Qualcomm/Adreno - // Started Version: 14 - // Ended Version: 53 - // Centroid in/out, used in the shaders, is used for multisample buffers to get the texel correctly - // When MSAA is disabled, it acts like a regular in/out - // Tends to cause the driver to render full white or black - BUG_BROKENCENTROID, - // Bug: INFO_LOG_LENGTH broken - // Affected devices: Qualcomm/Adreno - // Started Version: ? (Noticed on v14) - // Ended Version: 53 - // When compiling a shader, it is important that when it fails, - // you first get the length of the information log prior to grabbing it. - // This allows you to allocate an array to store all of the log - // Adreno devices /always/ return 0 when querying GL_INFO_LOG_LENGTH - // They also max out at 1024 bytes(1023 characters + null terminator) for the log BUG_BROKENINFOLOG, // Bug: UBO buffer offset broken // Affected devices: all mesa drivers @@ -104,22 +78,6 @@ namespace DriverDetails // Please see issue #6105 on Google Code. Let's hope buffer storage solves this issues. // TODO: Detect broken drivers. BUG_BROKENPINNEDMEMORY, - // Bug: Entirely broken UBOs - // Affected devices: Qualcomm/Adreno - // Started Version: ? (Noticed on v45) - // Ended Version: 53 - // Uniform buffers are entirely broken on Qualcomm drivers with v45 - // Trying to use the uniform buffers causes a malloc to fail inside the driver - // To be safe, blanket drivers from v41 - v45 - BUG_ANNIHILATEDUBOS, - // Bug : Can't draw on screen text and clear correctly. - // Affected devices: Qualcomm/Adreno - // Started Version: ? - // Ended Version: 53 - // Current code for drawing on screen text and clearing the framebuffer doesn't work on Adreno - // Drawing on screen text causes the whole screen to swizzle in a terrible fashion - // Clearing the framebuffer causes one to never see a frame. - BUG_BROKENSWAP, // Bug: glBufferSubData/glMapBufferRange stalls + OOM // Affected devices: Adreno a3xx/Mali-t6xx // Started Version: -1 @@ -128,12 +86,6 @@ namespace DriverDetails // The driver stalls in each instance no matter what you do // Apparently Mali and Adreno share code in this regard since it was wrote by the same person. BUG_BROKENBUFFERSTREAM, - // Bug: GLSL ES 3.0 textureSize causes abort - // Affected devices: Adreno a3xx - // Started Version: -1 (Noticed in v53) - // Ended Version: 66 - // If a shader includes a textureSize function call then the shader compiler will call abort() - BUG_BROKENTEXTURESIZE, // Bug: ARB_buffer_storage doesn't work with ARRAY_BUFFER type streams // Affected devices: GeForce 4xx+ // Started Version: -1 @@ -169,14 +121,6 @@ namespace DriverDetails // It works for all the buffer types we use except GL_ELEMENT_ARRAY_BUFFER. // Causes complete blackscreen issues. BUG_INTELBROKENBUFFERSTORAGE, - // Bug: Qualcomm has broken attributeless rendering - // Affected devices: Adreno - // Started Version: -1 - // Ended Version: v66 (07-09-2014 dev version), v95 shipping - // Qualcomm has had attributeless rendering broken forever - // This was fixed in a v66 development version, the first shipping driver version with the release was v95. - // To be safe, make v95 the minimum version to work around this issue - BUG_BROKENATTRIBUTELESS, // Bug: Qualcomm has broken boolean negation // Affected devices: Adreno // Started Version: -1 @@ -202,38 +146,30 @@ namespace DriverDetails // if (cond == false) BUG_BROKENNEGATEDBOOLEAN, - // Bug: Qualcomm has broken ivec to scalar and ivec to ivec bitshifts + // Bug: glCopyImageSubData doesn't work on i965 + // Started Version: -1 + // Ended Version: 10.6.4 + // Mesa meta misses to disable the scissor test. + BUG_BROKENCOPYIMAGE, + + // Bug: Qualcomm has broken OpenGL ES 3.1 support // Affected devices: Adreno // Started Version: -1 - // Ended Version: 46 (TODO: Test more devices, the real end is currently unknown) - // Qualcomm has broken integer vector to integer bitshifts, and integer vector to integer vector bitshifts - // A compilation error is generated when trying to compile the shaders. - // - // For example: - // Broken on Qualcomm: - // ivec4 ab = ivec4(1,1,1,1); - // ab <<= 2; - // - // Working on Qualcomm: - // ivec4 ab = ivec4(1,1,1,1); - // ab.x <<= 2; - // ab.y <<= 2; - // ab.z <<= 2; - // ab.w <<= 2; - // - // Broken on Qualcomm: - // ivec4 ab = ivec4(1,1,1,1); - // ivec4 cd = ivec4(1,2,3,4); - // ab <<= cd; - // - // Working on Qualcomm: - // ivec4 ab = ivec4(1,1,1,1); - // ivec4 cd = ivec4(1,2,3,4); - // ab.x <<= cd.x; - // ab.y <<= cd.y; - // ab.z <<= cd.z; - // ab.w <<= cd.w; - BUG_BROKENIVECSHIFTS, + // Ended Version: -1 + // This isn't fully researched, but at the very least Qualcomm doesn't implement Geometry shader features fully. + // Until each bug is fully investigated, just disable GLES 3.1 entirely on these devices. + BUG_BROKENGLES31, + + // Bug: ARM Mali managed to break disabling vsync + // Affected Devices: Mali + // Started Version: r5p0-rev2 + // Ended Version: -1 + // If we disable vsync with eglSwapInterval(dpy, 0) then the screen will stop showing new updates after a handful of swaps. + // This was noticed on a Samsung Galaxy S6 with its Android 5.1.1 update. + // The default Android 5.0 image didn't encounter this issue. + // We can't actually detect what the driver version is on Android, so until the driver version lands that displays the version in + // the GL_VERSION string, we will have to force vsync to be enabled at all times. + BUG_BROKENVSYNC, }; // Initializes our internal vendor, device family, and driver version diff --git a/Source/Core/VideoCommon/Fifo.cpp b/Source/Core/VideoCommon/Fifo.cpp index a788d61ce7..2395d2b324 100644 --- a/Source/Core/VideoCommon/Fifo.cpp +++ b/Source/Core/VideoCommon/Fifo.cpp @@ -217,7 +217,7 @@ static void ReadDataFromFifo(u32 readPtr) size_t existing_len = s_video_buffer_write_ptr - s_video_buffer_read_ptr; if (len > (size_t)(FIFO_SIZE - existing_len)) { - PanicAlert("FIFO out of bounds (existing %lu + new %lu > %lu)", (unsigned long) existing_len, (unsigned long) len, (unsigned long) FIFO_SIZE); + PanicAlert("FIFO out of bounds (existing %zu + new %zu > %lu)", existing_len, len, (unsigned long) FIFO_SIZE); return; } memmove(s_video_buffer, s_video_buffer_read_ptr, existing_len); @@ -254,7 +254,7 @@ static void ReadDataFromFifoOnCPU(u32 readPtr) size_t existing_len = write_ptr - s_video_buffer_pp_read_ptr; if (len > (size_t)(FIFO_SIZE - existing_len)) { - PanicAlert("FIFO out of bounds (existing %lu + new %lu > %lu)", (unsigned long) existing_len, (unsigned long) len, (unsigned long) FIFO_SIZE); + PanicAlert("FIFO out of bounds (existing %zu + new %zu > %lu)", existing_len, len, (unsigned long) FIFO_SIZE); return; } } diff --git a/Source/Core/VideoCommon/FramebufferManagerBase.cpp b/Source/Core/VideoCommon/FramebufferManagerBase.cpp index 10d1cfa414..7981dfce4d 100644 --- a/Source/Core/VideoCommon/FramebufferManagerBase.cpp +++ b/Source/Core/VideoCommon/FramebufferManagerBase.cpp @@ -2,7 +2,7 @@ // Licensed under GPLv2+ // Refer to the license.txt file included. - +#include <algorithm> #include "VideoCommon/FramebufferManagerBase.h" #include "VideoCommon/RenderBase.h" #include "VideoCommon/VideoConfig.h" @@ -115,17 +115,17 @@ const XFBSourceBase* const* FramebufferManagerBase::GetVirtualXFBSource(u32 xfbA return &m_overlappingXFBArray[0]; } -void FramebufferManagerBase::CopyToXFB(u32 xfbAddr, u32 fbWidth, u32 fbHeight, const EFBRectangle& sourceRc,float Gamma) +void FramebufferManagerBase::CopyToXFB(u32 xfbAddr, u32 fbStride, u32 fbHeight, const EFBRectangle& sourceRc, float Gamma) { if (g_ActiveConfig.bUseRealXFB) - g_framebuffer_manager->CopyToRealXFB(xfbAddr, fbWidth, fbHeight, sourceRc,Gamma); + g_framebuffer_manager->CopyToRealXFB(xfbAddr, fbStride, fbHeight, sourceRc, Gamma); else - CopyToVirtualXFB(xfbAddr, fbWidth, fbHeight, sourceRc,Gamma); + CopyToVirtualXFB(xfbAddr, fbStride, fbHeight, sourceRc, Gamma); } -void FramebufferManagerBase::CopyToVirtualXFB(u32 xfbAddr, u32 fbWidth, u32 fbHeight, const EFBRectangle& sourceRc,float Gamma) +void FramebufferManagerBase::CopyToVirtualXFB(u32 xfbAddr, u32 fbStride, u32 fbHeight, const EFBRectangle& sourceRc, float Gamma) { - VirtualXFBListType::iterator vxfb = FindVirtualXFB(xfbAddr, fbWidth, fbHeight); + VirtualXFBListType::iterator vxfb = FindVirtualXFB(xfbAddr, sourceRc.GetWidth(), fbHeight); if (m_virtualXFBList.end() == vxfb) { @@ -165,7 +165,7 @@ void FramebufferManagerBase::CopyToVirtualXFB(u32 xfbAddr, u32 fbWidth, u32 fbHe } vxfb->xfbSource->srcAddr = vxfb->xfbAddr = xfbAddr; - vxfb->xfbSource->srcWidth = vxfb->xfbWidth = fbWidth; + vxfb->xfbSource->srcWidth = vxfb->xfbWidth = sourceRc.GetWidth(); vxfb->xfbSource->srcHeight = vxfb->xfbHeight = fbHeight; vxfb->xfbSource->sourceRc = g_renderer->ConvertEFBRectangle(sourceRc); @@ -182,17 +182,12 @@ FramebufferManagerBase::VirtualXFBListType::iterator FramebufferManagerBase::Fin const u32 srcLower = xfbAddr; const u32 srcUpper = xfbAddr + 2 * width * height; - VirtualXFBListType::iterator it = m_virtualXFBList.begin(); - for (; it != m_virtualXFBList.end(); ++it) - { - const u32 dstLower = it->xfbAddr; - const u32 dstUpper = it->xfbAddr + 2 * it->xfbWidth * it->xfbHeight; - - if (dstLower >= srcLower && dstUpper <= srcUpper) - break; - } + return std::find_if(m_virtualXFBList.begin(), m_virtualXFBList.end(), [srcLower, srcUpper](const VirtualXFB& xfb) { + const u32 dstLower = xfb.xfbAddr; + const u32 dstUpper = xfb.xfbAddr + 2 * xfb.xfbWidth * xfb.xfbHeight; - return it; + return dstLower >= srcLower && dstUpper <= srcUpper; + }); } void FramebufferManagerBase::ReplaceVirtualXFB() diff --git a/Source/Core/VideoCommon/FramebufferManagerBase.h b/Source/Core/VideoCommon/FramebufferManagerBase.h index bd6360877c..6127037268 100644 --- a/Source/Core/VideoCommon/FramebufferManagerBase.h +++ b/Source/Core/VideoCommon/FramebufferManagerBase.h @@ -45,7 +45,7 @@ public: FramebufferManagerBase(); virtual ~FramebufferManagerBase(); - static void CopyToXFB(u32 xfbAddr, u32 fbWidth, u32 fbHeight, const EFBRectangle& sourceRc,float Gamma); + static void CopyToXFB(u32 xfbAddr, u32 fbStride, u32 fbHeight, const EFBRectangle& sourceRc,float Gamma); static const XFBSourceBase* const* GetXFBSource(u32 xfbAddr, u32 fbWidth, u32 fbHeight, u32* xfbCount); static void SetLastXfbWidth(unsigned int width) { s_last_xfb_width = width; } @@ -87,7 +87,7 @@ private: static void ReplaceVirtualXFB(); // TODO: merge these virtual funcs, they are nearly all the same - virtual void CopyToRealXFB(u32 xfbAddr, u32 fbWidth, u32 fbHeight, const EFBRectangle& sourceRc,float Gamma = 1.0f) = 0; + virtual void CopyToRealXFB(u32 xfbAddr, u32 fbStride, u32 fbHeight, const EFBRectangle& sourceRc,float Gamma = 1.0f) = 0; static void CopyToVirtualXFB(u32 xfbAddr, u32 fbWidth, u32 fbHeight, const EFBRectangle& sourceRc,float Gamma = 1.0f); static const XFBSourceBase* const* GetRealXFBSource(u32 xfbAddr, u32 fbWidth, u32 fbHeight, u32* xfbCount); diff --git a/Source/Core/VideoCommon/GeometryShaderGen.cpp b/Source/Core/VideoCommon/GeometryShaderGen.cpp index 5ec98b3813..f36aa94a9a 100644 --- a/Source/Core/VideoCommon/GeometryShaderGen.cpp +++ b/Source/Core/VideoCommon/GeometryShaderGen.cpp @@ -94,11 +94,11 @@ static inline void GenerateGeometryShader(T& out, u32 primitive_type, API_TYPE A out.Write("#define InstanceID gl_InvocationID\n"); out.Write("in VertexData {\n"); - GenerateVSOutputMembers<T>(out, ApiType, g_ActiveConfig.backend_info.bSupportsBindingLayout ? "centroid" : "centroid in"); + GenerateVSOutputMembers<T>(out, ApiType, GetInterpolationQualifier(ApiType, true, true)); out.Write("} vs[%d];\n", vertex_in); out.Write("out VertexData {\n"); - GenerateVSOutputMembers<T>(out, ApiType, g_ActiveConfig.backend_info.bSupportsBindingLayout ? "centroid" : "centroid out"); + GenerateVSOutputMembers<T>(out, ApiType, GetInterpolationQualifier(ApiType, false, true)); if (g_ActiveConfig.iStereoMode > 0) out.Write("\tflat int layer;\n"); diff --git a/Source/Core/VideoCommon/HiresTextures.cpp b/Source/Core/VideoCommon/HiresTextures.cpp index efe3ab58bc..e6f687a551 100644 --- a/Source/Core/VideoCommon/HiresTextures.cpp +++ b/Source/Core/VideoCommon/HiresTextures.cpp @@ -140,7 +140,10 @@ void HiresTexture::Prefetch() Common::SetCurrentThreadName("Prefetcher"); size_t size_sum = 0; - size_t max_mem = MemPhysical() / 2; + size_t sys_mem = MemPhysical(); + size_t recommended_min_mem = 2 * size_t(1024 * 1024 * 1024); + // keep 2GB memory for system stability if system RAM is 4GB+ - use half of memory in other cases + size_t max_mem = (sys_mem / 2 < recommended_min_mem) ? (sys_mem / 2) : (sys_mem - recommended_min_mem); u32 starttime = Common::Timer::GetTimeMs(); for (const auto& entry : s_textureMap) { diff --git a/Source/Core/VideoCommon/ImageWrite.cpp b/Source/Core/VideoCommon/ImageWrite.cpp index 38ca509c3e..f50c5d02b4 100644 --- a/Source/Core/VideoCommon/ImageWrite.cpp +++ b/Source/Core/VideoCommon/ImageWrite.cpp @@ -8,6 +8,8 @@ #include "png.h" #include "Common/FileUtil.h" +#include "Common/MsgHandler.h" +#include "Common/Logging/Log.h" #include "VideoCommon/ImageWrite.h" bool SaveData(const std::string& filename, const char* data) diff --git a/Source/Core/VideoCommon/IndexGenerator.cpp b/Source/Core/VideoCommon/IndexGenerator.cpp index 19a6642f66..c1941c09b8 100644 --- a/Source/Core/VideoCommon/IndexGenerator.cpp +++ b/Source/Core/VideoCommon/IndexGenerator.cpp @@ -4,6 +4,7 @@ #include <cstddef> +#include "Common/Common.h" #include "Common/CommonTypes.h" #include "VideoCommon/IndexGenerator.h" #include "VideoCommon/OpcodeDecoding.h" diff --git a/Source/Core/VideoCommon/LightingShaderGen.h b/Source/Core/VideoCommon/LightingShaderGen.h index 83ecfb4174..ae388ac5ed 100644 --- a/Source/Core/VideoCommon/LightingShaderGen.h +++ b/Source/Core/VideoCommon/LightingShaderGen.h @@ -253,10 +253,7 @@ static void GenerateLightingShader(T& object, LightingUidData& uid_data, int com } } object.Write("lacc = clamp(lacc, 0, 255);\n"); - if (DriverDetails::HasBug(DriverDetails::BUG_BROKENIVECSHIFTS)) - object.Write("%s%d = float4(irshift((mat * (lacc + irshift(lacc, 7))), 8)) / 255.0;\n", dest, j); - else - object.Write("%s%d = float4((mat * (lacc + (lacc >> 7))) >> 8) / 255.0;\n", dest, j); + object.Write("%s%d = float4((mat * (lacc + (lacc >> 7))) >> 8) / 255.0;\n", dest, j); object.Write("}\n"); } } diff --git a/Source/Core/VideoCommon/LookUpTables.h b/Source/Core/VideoCommon/LookUpTables.h index d821597c8f..4bda199be5 100644 --- a/Source/Core/VideoCommon/LookUpTables.h +++ b/Source/Core/VideoCommon/LookUpTables.h @@ -6,25 +6,25 @@ #include "Common/CommonTypes.h" -inline u8 Convert3To8(u8 v) +constexpr u8 Convert3To8(u8 v) { // Swizzle bits: 00000123 -> 12312312 return (v << 5) | (v << 2) | (v >> 1); } -inline u8 Convert4To8(u8 v) +constexpr u8 Convert4To8(u8 v) { // Swizzle bits: 00001234 -> 12341234 return (v << 4) | v; } -inline u8 Convert5To8(u8 v) +constexpr u8 Convert5To8(u8 v) { // Swizzle bits: 00012345 -> 12345123 return (v << 3) | (v >> 2); } -inline u8 Convert6To8(u8 v) +constexpr u8 Convert6To8(u8 v) { // Swizzle bits: 00123456 -> 12345612 return (v << 2) | (v >> 4); diff --git a/Source/Core/VideoCommon/NativeVertexFormat.h b/Source/Core/VideoCommon/NativeVertexFormat.h index c64d94fa77..df86b90946 100644 --- a/Source/Core/VideoCommon/NativeVertexFormat.h +++ b/Source/Core/VideoCommon/NativeVertexFormat.h @@ -4,10 +4,12 @@ #pragma once +#include <cstring> #include <functional> // for hash -#include "Common/Common.h" +#include "Common/CommonTypes.h" #include "Common/Hash.h" +#include "Common/NonCopyable.h" // m_components enum diff --git a/Source/Core/VideoCommon/OnScreenDisplay.cpp b/Source/Core/VideoCommon/OnScreenDisplay.cpp index ddd80065e1..dfd597f0ed 100644 --- a/Source/Core/VideoCommon/OnScreenDisplay.cpp +++ b/Source/Core/VideoCommon/OnScreenDisplay.cpp @@ -71,7 +71,7 @@ void ClearMessages() // On-Screen Display Callbacks void AddCallback(CallbackType type, Callback cb) { - s_callbacks.insert(std::pair<CallbackType, Callback>(type, cb)); + s_callbacks.emplace(type, cb); } void DoCallbacks(CallbackType type) diff --git a/Source/Core/VideoCommon/OpcodeDecoding.h b/Source/Core/VideoCommon/OpcodeDecoding.h index ec1b2652e4..a79bf54be2 100644 --- a/Source/Core/VideoCommon/OpcodeDecoding.h +++ b/Source/Core/VideoCommon/OpcodeDecoding.h @@ -38,8 +38,6 @@ #define GX_DRAW_LINE_STRIP 0x6 // 0xB0 #define GX_DRAW_POINTS 0x7 // 0xB8 -extern bool g_bRecordFifoData; - void OpcodeDecoder_Init(); void OpcodeDecoder_Shutdown(); diff --git a/Source/Core/VideoCommon/PixelShaderGen.cpp b/Source/Core/VideoCommon/PixelShaderGen.cpp index ad251203f9..2a65a32836 100644 --- a/Source/Core/VideoCommon/PixelShaderGen.cpp +++ b/Source/Core/VideoCommon/PixelShaderGen.cpp @@ -6,6 +6,7 @@ #include <cmath> #include <cstdio> +#include "Common/Common.h" #include "VideoCommon/BoundingBox.h" #include "VideoCommon/BPMemory.h" #include "VideoCommon/ConstantManager.h" @@ -17,6 +18,24 @@ #include "VideoCommon/VideoConfig.h" #include "VideoCommon/XFMemory.h" // for texture projection mode +// TODO: Get rid of these +enum : u32 +{ + C_COLORMATRIX = 0, // 0 + C_COLORS = 0, // 0 + C_KCOLORS = C_COLORS + 4, // 4 + C_ALPHA = C_KCOLORS + 4, // 8 + C_TEXDIMS = C_ALPHA + 1, // 9 + C_ZBIAS = C_TEXDIMS + 8, // 17 + C_INDTEXSCALE = C_ZBIAS + 2, // 19 + C_INDTEXMTX = C_INDTEXSCALE + 2, // 21 + C_FOGCOLOR = C_INDTEXMTX + 6, // 27 + C_FOGI = C_FOGCOLOR + 1, // 28 + C_FOGF = C_FOGI + 1, // 29 + C_ZSLOPE = C_FOGF + 2, // 31 + C_EFBSCALE = C_ZSLOPE + 1, // 32 + C_PENVCONST_END = C_EFBSCALE + 1 +}; static const char *tevKSelTableC[] = { @@ -137,7 +156,7 @@ static const char *tevRasTable[] = static const char *tevCOutputTable[] = { "prev.rgb", "c0.rgb", "c1.rgb", "c2.rgb" }; static const char *tevAOutputTable[] = { "prev.a", "c0.a", "c1.a", "c2.a" }; -static char text[16384]; +static char text[32768]; template<class T> static inline void WriteStage(T& out, pixel_shader_uid_data* uid_data, int n, API_TYPE ApiType, const char swapModeTable[4][5]); template<class T> static inline void WriteTevRegular(T& out, const char* components, int bias, int op, int clamp, int shift); @@ -196,30 +215,6 @@ static inline void GeneratePixelShader(T& out, DSTALPHA_MODE dstAlphaMode, API_T "int3 itrunc(float3 x) { return int3(trunc(x)); }\n" "int4 itrunc(float4 x) { return int4(trunc(x)); }\n\n"); - if (DriverDetails::HasBug(DriverDetails::BUG_BROKENIVECSHIFTS)) - { - // Add functions to do shifts on scalars and ivecs. - // These functions all have the same name to enable them to be used no matter what code is generated. - // For example: tev color op code uses .rgb as a swizzle, but alpha code only uses .a. - out.Write("int ilshift(int a, int b) { return a << b; }\n" - "int irshift(int a, int b) { return a >> b; }\n" - - "int2 ilshift(int2 a, int2 b) { return int2(a.x << b.x, a.y << b.y); }\n" - "int2 ilshift(int2 a, int b) { return int2(a.x << b, a.y << b); }\n" - "int2 irshift(int2 a, int2 b) { return int2(a.x >> b.x, a.y >> b.y); }\n" - "int2 irshift(int2 a, int b) { return int2(a.x >> b, a.y >> b); }\n" - - "int3 ilshift(int3 a, int3 b) { return int3(a.x << b.x, a.y << b.y, a.z << b.z); }\n" - "int3 ilshift(int3 a, int b) { return int3(a.x << b, a.y << b, a.z << b); }\n" - "int3 irshift(int3 a, int3 b) { return int3(a.x >> b.x, a.y >> b.y, a.z >> b.z); }\n" - "int3 irshift(int3 a, int b) { return int3(a.x >> b, a.y >> b, a.z >> b); }\n" - - "int4 ilshift(int4 a, int4 b) { return int4(a.x << b.x, a.y << b.y, a.z << b.z, a.w << b.w); }\n" - "int4 ilshift(int4 a, int b) { return int4(a.x << b, a.y << b, a.z << b, a.w << b); }\n" - "int4 irshift(int4 a, int4 b) { return int4(a.x >> b.x, a.y >> b.y, a.z >> b.z, a.w >> b.w); }\n" - "int4 irshift(int4 a, int b) { return int4(a.x >> b, a.y >> b, a.z >> b, a.w >> b); }\n\n"); - } - if (ApiType == API_OPENGL) { // Declare samplers @@ -338,6 +333,8 @@ static inline void GeneratePixelShader(T& out, DSTALPHA_MODE dstAlphaMode, API_T warn_once = false; } + uid_data->msaa = g_ActiveConfig.iMultisampleMode > 0; + uid_data->ssaa = g_ActiveConfig.iMultisampleMode > 0 && g_ActiveConfig.bSSAA; if (ApiType == API_OPENGL) { out.Write("out vec4 ocol0;\n"); @@ -347,19 +344,11 @@ static inline void GeneratePixelShader(T& out, DSTALPHA_MODE dstAlphaMode, API_T if (per_pixel_depth) out.Write("#define depth gl_FragDepth\n"); - // We use the flag "centroid" to fix some MSAA rendering bugs. With MSAA, the - // pixel shader will be executed for each pixel which has at least one passed sample. - // So there may be rendered pixels where the center of the pixel isn't in the primitive. - // As the pixel shader usually renders at the center of the pixel, this position may be - // outside the primitive. This will lead to sampling outside the texture, sign changes, ... - // As a workaround, we interpolate at the centroid of the coveraged pixel, which - // is always inside the primitive. - // Without MSAA, this flag is defined to have no effect. uid_data->stereo = g_ActiveConfig.iStereoMode > 0; if (g_ActiveConfig.backend_info.bSupportsGeometryShaders) { out.Write("in VertexData {\n"); - GenerateVSOutputMembers<T>(out, ApiType, g_ActiveConfig.backend_info.bSupportsBindingLayout ? "centroid" : "centroid in"); + GenerateVSOutputMembers<T>(out, ApiType, GetInterpolationQualifier(ApiType, true, true)); if (g_ActiveConfig.iStereoMode > 0) out.Write("\tflat int layer;\n"); @@ -368,19 +357,19 @@ static inline void GeneratePixelShader(T& out, DSTALPHA_MODE dstAlphaMode, API_T } else { - out.Write("centroid in float4 colors_0;\n"); - out.Write("centroid in float4 colors_1;\n"); + out.Write("%s in float4 colors_0;\n", GetInterpolationQualifier(ApiType)); + out.Write("%s in float4 colors_1;\n", GetInterpolationQualifier(ApiType)); // compute window position if needed because binding semantic WPOS is not widely supported // Let's set up attributes for (unsigned int i = 0; i < numTexgen; ++i) { - out.Write("centroid in float3 uv%d;\n", i); + out.Write("%s in float3 uv%d;\n", GetInterpolationQualifier(ApiType), i); } - out.Write("centroid in float4 clipPos;\n"); + out.Write("%s in float4 clipPos;\n", GetInterpolationQualifier(ApiType)); if (g_ActiveConfig.bEnablePixelLighting) { - out.Write("centroid in float3 Normal;\n"); - out.Write("centroid in float3 WorldPos;\n"); + out.Write("%s in float3 Normal;\n", GetInterpolationQualifier(ApiType)); + out.Write("%s in float3 WorldPos;\n", GetInterpolationQualifier(ApiType)); } } @@ -401,17 +390,17 @@ static inline void GeneratePixelShader(T& out, DSTALPHA_MODE dstAlphaMode, API_T dstAlphaMode == DSTALPHA_DUAL_SOURCE_BLEND ? "\n out float4 ocol1 : SV_Target1," : "", per_pixel_depth ? "\n out float depth : SV_Depth," : ""); - out.Write(" in centroid float4 colors_0 : COLOR0,\n"); - out.Write(" in centroid float4 colors_1 : COLOR1\n"); + out.Write(" in %s float4 colors_0 : COLOR0,\n", GetInterpolationQualifier(ApiType)); + out.Write(" in %s float4 colors_1 : COLOR1\n", GetInterpolationQualifier(ApiType)); // compute window position if needed because binding semantic WPOS is not widely supported for (unsigned int i = 0; i < numTexgen; ++i) - out.Write(",\n in centroid float3 uv%d : TEXCOORD%d", i, i); - out.Write(",\n in centroid float4 clipPos : TEXCOORD%d", numTexgen); + out.Write(",\n in %s float3 uv%d : TEXCOORD%d", GetInterpolationQualifier(ApiType), i, i); + out.Write(",\n in %s float4 clipPos : TEXCOORD%d", GetInterpolationQualifier(ApiType), numTexgen); if (g_ActiveConfig.bEnablePixelLighting) { - out.Write(",\n in centroid float3 Normal : TEXCOORD%d", numTexgen + 1); - out.Write(",\n in centroid float3 WorldPos : TEXCOORD%d", numTexgen + 2); + out.Write(",\n in %s float3 Normal : TEXCOORD%d", GetInterpolationQualifier(ApiType), numTexgen + 1); + out.Write(",\n in %s float3 WorldPos : TEXCOORD%d", GetInterpolationQualifier(ApiType), numTexgen + 2); } uid_data->stereo = g_ActiveConfig.iStereoMode > 0; if (g_ActiveConfig.iStereoMode > 0) @@ -499,11 +488,7 @@ static inline void GeneratePixelShader(T& out, DSTALPHA_MODE dstAlphaMode, API_T if (texcoord < numTexgen) { out.SetConstantsUsed(C_INDTEXSCALE+i/2,C_INDTEXSCALE+i/2); - - if (DriverDetails::HasBug(DriverDetails::BUG_BROKENIVECSHIFTS)) - out.Write("\ttempcoord = irshift(fixpoint_uv%d, " I_INDTEXSCALE"[%d].%s);\n", texcoord, i / 2, (i & 1) ? "zw" : "xy"); - else - out.Write("\ttempcoord = fixpoint_uv%d >> " I_INDTEXSCALE"[%d].%s;\n", texcoord, i / 2, (i & 1) ? "zw" : "xy"); + out.Write("\ttempcoord = fixpoint_uv%d >> " I_INDTEXSCALE"[%d].%s;\n", texcoord, i / 2, (i & 1) ? "zw" : "xy"); } else out.Write("\ttempcoord = int2(0, 0);\n"); @@ -729,22 +714,11 @@ static inline void WriteStage(T& out, pixel_shader_uid_data* uid_data, int n, AP int mtxidx = 2*(bpmem.tevind[n].mid-1); out.SetConstantsUsed(C_INDTEXMTX+mtxidx, C_INDTEXMTX+mtxidx); - if (DriverDetails::HasBug(DriverDetails::BUG_BROKENIVECSHIFTS)) - { - out.Write("\tint2 indtevtrans%d = irshift(int2(idot(" I_INDTEXMTX"[%d].xyz, iindtevcrd%d), idot(" I_INDTEXMTX"[%d].xyz, iindtevcrd%d)), 3);\n", n, mtxidx, n, mtxidx+1, n); - - // TODO: should use a shader uid branch for this for better performance - out.Write("\tif (" I_INDTEXMTX"[%d].w >= 0) indtevtrans%d = irshift(indtevtrans%d, " I_INDTEXMTX"[%d].w);\n", mtxidx, n, n, mtxidx); - out.Write("\telse indtevtrans%d = ilshift(indtevtrans%d, -" I_INDTEXMTX"[%d].w);\n", n, n, mtxidx); - } - else - { - out.Write("\tint2 indtevtrans%d = int2(idot(" I_INDTEXMTX"[%d].xyz, iindtevcrd%d), idot(" I_INDTEXMTX"[%d].xyz, iindtevcrd%d)) >> 3;\n", n, mtxidx, n, mtxidx+1, n); - - // TODO: should use a shader uid branch for this for better performance - out.Write("\tif (" I_INDTEXMTX"[%d].w >= 0) indtevtrans%d = indtevtrans%d >> " I_INDTEXMTX"[%d].w;\n", mtxidx, n, n, mtxidx); - out.Write("\telse indtevtrans%d = indtevtrans%d << (-" I_INDTEXMTX"[%d].w);\n", n, n, mtxidx); - } + out.Write("\tint2 indtevtrans%d = int2(idot(" I_INDTEXMTX"[%d].xyz, iindtevcrd%d), idot(" I_INDTEXMTX"[%d].xyz, iindtevcrd%d)) >> 3;\n", n, mtxidx, n, mtxidx+1, n); + + // TODO: should use a shader uid branch for this for better performance + out.Write("\tif (" I_INDTEXMTX"[%d].w >= 0) indtevtrans%d = indtevtrans%d >> " I_INDTEXMTX"[%d].w;\n", mtxidx, n, n, mtxidx); + out.Write("\telse indtevtrans%d = indtevtrans%d << (-" I_INDTEXMTX"[%d].w);\n", n, n, mtxidx); } else if (bpmem.tevind[n].mid <= 7 && bHasTexCoord) { // s matrix @@ -752,20 +726,10 @@ static inline void WriteStage(T& out, pixel_shader_uid_data* uid_data, int n, AP int mtxidx = 2*(bpmem.tevind[n].mid-5); out.SetConstantsUsed(C_INDTEXMTX+mtxidx, C_INDTEXMTX+mtxidx); - if (DriverDetails::HasBug(DriverDetails::BUG_BROKENIVECSHIFTS)) - { - out.Write("\tint2 indtevtrans%d = irshift(int2(fixpoint_uv%d * iindtevcrd%d.xx), 8);\n", n, texcoord, n); - - out.Write("\tif (" I_INDTEXMTX"[%d].w >= 0) indtevtrans%d = irshift(indtevtrans%d, " I_INDTEXMTX"[%d].w);\n", mtxidx, n, n, mtxidx); - out.Write("\telse indtevtrans%d = ilshift(indtevtrans%d, -" I_INDTEXMTX"[%d].w);\n", n, n, mtxidx); - } - else - { - out.Write("\tint2 indtevtrans%d = int2(fixpoint_uv%d * iindtevcrd%d.xx) >> 8;\n", n, texcoord, n); + out.Write("\tint2 indtevtrans%d = int2(fixpoint_uv%d * iindtevcrd%d.xx) >> 8;\n", n, texcoord, n); - out.Write("\tif (" I_INDTEXMTX"[%d].w >= 0) indtevtrans%d = indtevtrans%d >> " I_INDTEXMTX"[%d].w;\n", mtxidx, n, n, mtxidx); - out.Write("\telse indtevtrans%d = indtevtrans%d << (-" I_INDTEXMTX"[%d].w);\n", n, n, mtxidx); - } + out.Write("\tif (" I_INDTEXMTX"[%d].w >= 0) indtevtrans%d = indtevtrans%d >> " I_INDTEXMTX"[%d].w;\n", mtxidx, n, n, mtxidx); + out.Write("\telse indtevtrans%d = indtevtrans%d << (-" I_INDTEXMTX"[%d].w);\n", n, n, mtxidx); } else if (bpmem.tevind[n].mid <= 11 && bHasTexCoord) { // t matrix @@ -773,20 +737,10 @@ static inline void WriteStage(T& out, pixel_shader_uid_data* uid_data, int n, AP int mtxidx = 2*(bpmem.tevind[n].mid-9); out.SetConstantsUsed(C_INDTEXMTX+mtxidx, C_INDTEXMTX+mtxidx); - if (DriverDetails::HasBug(DriverDetails::BUG_BROKENIVECSHIFTS)) - { - out.Write("\tint2 indtevtrans%d = irshift(int2(fixpoint_uv%d * iindtevcrd%d.yy), 8);\n", n, texcoord, n); + out.Write("\tint2 indtevtrans%d = int2(fixpoint_uv%d * iindtevcrd%d.yy) >> 8;\n", n, texcoord, n); - out.Write("\tif (" I_INDTEXMTX"[%d].w >= 0) indtevtrans%d = irshift(indtevtrans%d, " I_INDTEXMTX"[%d].w);\n", mtxidx, n, n, mtxidx); - out.Write("\telse indtevtrans%d = ilshift(indtevtrans%d, -" I_INDTEXMTX"[%d].w);\n", n, n, mtxidx); - } - else - { - out.Write("\tint2 indtevtrans%d = int2(fixpoint_uv%d * iindtevcrd%d.yy) >> 8;\n", n, texcoord, n); - - out.Write("\tif (" I_INDTEXMTX"[%d].w >= 0) indtevtrans%d = indtevtrans%d >> " I_INDTEXMTX"[%d].w;\n", mtxidx, n, n, mtxidx); - out.Write("\telse indtevtrans%d = indtevtrans%d << (-" I_INDTEXMTX"[%d].w);\n", n, n, mtxidx); - } + out.Write("\tif (" I_INDTEXMTX"[%d].w >= 0) indtevtrans%d = indtevtrans%d >> " I_INDTEXMTX"[%d].w;\n", mtxidx, n, n, mtxidx); + out.Write("\telse indtevtrans%d = indtevtrans%d << (-" I_INDTEXMTX"[%d].w);\n", n, n, mtxidx); } else { @@ -825,10 +779,7 @@ static inline void WriteStage(T& out, pixel_shader_uid_data* uid_data, int n, AP out.Write("\ttevcoord.xy = wrappedcoord + indtevtrans%d;\n", n); // Emulate s24 overflows - if (DriverDetails::HasBug(DriverDetails::BUG_BROKENIVECSHIFTS)) - out.Write("\ttevcoord.xy = irshift(ilshift(tevcoord.xy, 8), 8);\n"); - else - out.Write("\ttevcoord.xy = (tevcoord.xy << 8) >> 8;\n"); + out.Write("\ttevcoord.xy = (tevcoord.xy << 8) >> 8;\n"); } TevStageCombiner::ColorCombiner &cc = bpmem.combiners[n].colorC; @@ -923,14 +874,14 @@ static inline void WriteStage(T& out, pixel_shader_uid_data* uid_data, int n, AP out.SetConstantsUsed(C_COLORS+ac.dest, C_COLORS+ac.dest); - out.Write("\ttevin_a = int4(%s, %s)&255;\n", tevCInputTable[cc.a], tevAInputTable[ac.a]); - out.Write("\ttevin_b = int4(%s, %s)&255;\n", tevCInputTable[cc.b], tevAInputTable[ac.b]); - out.Write("\ttevin_c = int4(%s, %s)&255;\n", tevCInputTable[cc.c], tevAInputTable[ac.c]); + out.Write("\ttevin_a = int4(%s, %s)&int4(255, 255, 255, 255);\n", tevCInputTable[cc.a], tevAInputTable[ac.a]); + out.Write("\ttevin_b = int4(%s, %s)&int4(255, 255, 255, 255);\n", tevCInputTable[cc.b], tevAInputTable[ac.b]); + out.Write("\ttevin_c = int4(%s, %s)&int4(255, 255, 255, 255);\n", tevCInputTable[cc.c], tevAInputTable[ac.c]); out.Write("\ttevin_d = int4(%s, %s);\n", tevCInputTable[cc.d], tevAInputTable[ac.d]); out.Write("\t// color combine\n"); out.Write("\t%s = clamp(", tevCOutputTable[cc.dest]); - if (cc.bias != TevBias_COMPARE) + if (cc.bias != TEVBIAS_COMPARE) { WriteTevRegular(out, "rgb", cc.bias, cc.op, cc.clamp, cc.shift); } @@ -960,7 +911,7 @@ static inline void WriteStage(T& out, pixel_shader_uid_data* uid_data, int n, AP out.Write("\t// alpha combine\n"); out.Write("\t%s = clamp(", tevAOutputTable[ac.dest]); - if (ac.bias != TevBias_COMPARE) + if (ac.bias != TEVBIAS_COMPARE) { WriteTevRegular(out, "a", ac.bias, ac.op, ac.clamp, ac.shift); } @@ -1035,37 +986,12 @@ static inline void WriteTevRegular(T& out, const char* components, int bias, int // - c is scaled from 0..255 to 0..256, which allows dividing the result by 256 instead of 255 // - if scale is bigger than one, it is moved inside the lerp calculation for increased accuracy // - a rounding bias is added before dividing by 256 - - if (DriverDetails::HasBug(DriverDetails::BUG_BROKENIVECSHIFTS)) - { - // Haxx - cleaner code by not having irshift and ilshift in the emitted code by omitting them if not used. - const char* leftShift = tevScaleTableLeft[shift]; - const char* rightShift = tevScaleTableRight[shift]; - - if (rightShift[0]) - out.Write("irshift(((tevin_d.%s%s)%s)", components, tevBiasTable[bias], tevScaleTableLeft[shift]); - else - out.Write("((tevin_d.%s%s)%s)", components, tevBiasTable[bias], tevScaleTableLeft[shift]); - out.Write(" %s ", tevOpTable[op]); - if (leftShift[0]) - out.Write("irshift((ilshift((ilshift(tevin_a.%s, 8) + (tevin_b.%s-tevin_a.%s)*(tevin_c.%s+irshift(tevin_c.%s, 7))), %s)%s), 8)", - components, components, components, components, components, - leftShift+4, tevLerpBias[2*op+(shift!=3)]); - else - out.Write("irshift(((ilshift(tevin_a.%s, 8) + (tevin_b.%s-tevin_a.%s)*(tevin_c.%s+irshift(tevin_c.%s, 7)))%s), 8)", - components, components, components, components, components, tevLerpBias[2*op+(shift!=3)]); - if (rightShift[0]) - out.Write(", %s)", rightShift+4); - } - else - { - out.Write("(((tevin_d.%s%s)%s)", components, tevBiasTable[bias], tevScaleTableLeft[shift]); - out.Write(" %s ", tevOpTable[op]); - out.Write("(((((tevin_a.%s<<8) + (tevin_b.%s-tevin_a.%s)*(tevin_c.%s+(tevin_c.%s>>7)))%s)%s)>>8)", - components, components, components, components, components, - tevScaleTableLeft[shift], tevLerpBias[2*op+(shift!=3)]); - out.Write(")%s", tevScaleTableRight[shift]); - } + out.Write("(((tevin_d.%s%s)%s)", components, tevBiasTable[bias], tevScaleTableLeft[shift]); + out.Write(" %s ", tevOpTable[op]); + out.Write("(((((tevin_a.%s<<8) + (tevin_b.%s-tevin_a.%s)*(tevin_c.%s+(tevin_c.%s>>7)))%s)%s)>>8)", + components, components, components, components, components, + tevScaleTableLeft[shift], tevLerpBias[2*op+(shift!=3)]); + out.Write(")%s", tevScaleTableRight[shift]); } template<class T> @@ -1081,14 +1007,14 @@ static inline void SampleTexture(T& out, const char *texcoords, const char *texs static const char *tevAlphaFuncsTable[] = { - "(false)", // NEVER - "(prev.a < %s)", // LESS - "(prev.a == %s)", // EQUAL - "(prev.a <= %s)", // LEQUAL - "(prev.a > %s)", // GREATER - "(prev.a != %s)", // NEQUAL - "(prev.a >= %s)", // GEQUAL - "(true)" // ALWAYS + "(false)", // NEVER + "(prev.a < %s)", // LESS + "(prev.a == %s)", // EQUAL + "(prev.a <= %s)", // LEQUAL + "(prev.a > %s)", // GREATER + "(prev.a != %s)", // NEQUAL + "(prev.a >= %s)", // GEQUAL + "(true)" // ALWAYS }; static const char *tevAlphaFunclogicTable[] = @@ -1228,10 +1154,7 @@ static inline void WriteFog(T& out, pixel_shader_uid_data* uid_data) } out.Write("\tint ifog = iround(fog * 256.0);\n"); - if (DriverDetails::HasBug(DriverDetails::BUG_BROKENIVECSHIFTS)) - out.Write("\tprev.rgb = irshift((prev.rgb * (256 - ifog) + " I_FOGCOLOR".rgb * ifog), 8);\n"); - else - out.Write("\tprev.rgb = (prev.rgb * (256 - ifog) + " I_FOGCOLOR".rgb * ifog) >> 8;\n"); + out.Write("\tprev.rgb = (prev.rgb * (256 - ifog) + " I_FOGCOLOR".rgb * ifog) >> 8;\n"); } void GetPixelShaderUid(PixelShaderUid& object, DSTALPHA_MODE dstAlphaMode, API_TYPE ApiType, u32 components) diff --git a/Source/Core/VideoCommon/PixelShaderGen.h b/Source/Core/VideoCommon/PixelShaderGen.h index 5a871fce6d..518207f7bc 100644 --- a/Source/Core/VideoCommon/PixelShaderGen.h +++ b/Source/Core/VideoCommon/PixelShaderGen.h @@ -9,23 +9,6 @@ #include "VideoCommon/ShaderGenCommon.h" #include "VideoCommon/VideoCommon.h" -// TODO: get rid of them as they aren't used -#define C_COLORMATRIX 0 // 0 -#define C_COLORS 0 // 0 -#define C_KCOLORS (C_COLORS + 4) // 4 -#define C_ALPHA (C_KCOLORS + 4) // 8 -#define C_TEXDIMS (C_ALPHA + 1) // 9 -#define C_ZBIAS (C_TEXDIMS + 8) //17 -#define C_INDTEXSCALE (C_ZBIAS + 2) //19 -#define C_INDTEXMTX (C_INDTEXSCALE + 2) //21 -#define C_FOGCOLOR (C_INDTEXMTX + 6) //27 -#define C_FOGI (C_FOGCOLOR + 1) //28 -#define C_FOGF (C_FOGI + 1) //29 -#define C_ZSLOPE (C_FOGF + 2) //31 -#define C_EFBSCALE (C_ZSLOPE + 1) //32 - -#define C_PENVCONST_END (C_EFBSCALE + 1) - // Different ways to achieve rendering with destination alpha enum DSTALPHA_MODE { @@ -65,9 +48,11 @@ struct pixel_shader_uid_data u32 early_ztest : 1; u32 bounding_box : 1; - // TODO: 31 bits of padding is a waste. Can we free up some bits elseware? + // TODO: 29 bits of padding is a waste. Can we free up some bits elseware? u32 zfreeze : 1; - u32 pad : 31; + u32 msaa : 1; + u32 ssaa : 1; + u32 pad : 29; u32 texMtxInfo_n_projection : 8; // 8x1 bit u32 tevindref_bi0 : 3; diff --git a/Source/Core/VideoCommon/PixelShaderManager.h b/Source/Core/VideoCommon/PixelShaderManager.h index 149b9cc55b..b3a7e8419b 100644 --- a/Source/Core/VideoCommon/PixelShaderManager.h +++ b/Source/Core/VideoCommon/PixelShaderManager.h @@ -39,7 +39,6 @@ public: static void SetEfbScaleChanged(); static void SetZSlope(float dfdx, float dfdy, float f0); static void SetIndMatrixChanged(int matrixidx); - static void SetTevKSelChanged(int id); static void SetZTextureTypeChanged(); static void SetIndTexScaleChanged(bool high); static void SetTexCoordChanged(u8 texmapid); diff --git a/Source/Core/VideoCommon/RenderBase.cpp b/Source/Core/VideoCommon/RenderBase.cpp index a688c86792..55a6cb8760 100644 --- a/Source/Core/VideoCommon/RenderBase.cpp +++ b/Source/Core/VideoCommon/RenderBase.cpp @@ -28,6 +28,8 @@ #include "Core/Movie.h" #include "Core/FifoPlayer/FifoRecorder.h" +#include "Core/HW/VideoInterface.h" + #include "VideoCommon/AVIDump.h" #include "VideoCommon/BPMemory.h" #include "VideoCommon/CommandProcessor.h" @@ -112,22 +114,23 @@ Renderer::~Renderer() #endif } -void Renderer::RenderToXFB(u32 xfbAddr, const EFBRectangle& sourceRc, u32 fbWidth, u32 fbHeight, float Gamma) +void Renderer::RenderToXFB(u32 xfbAddr, const EFBRectangle& sourceRc, u32 fbStride, u32 fbHeight, float Gamma) { CheckFifoRecording(); - if (!fbWidth || !fbHeight) + if (!fbStride || !fbHeight) return; XFBWrited = true; if (g_ActiveConfig.bUseXFB) { - FramebufferManagerBase::CopyToXFB(xfbAddr, fbWidth, fbHeight, sourceRc, Gamma); + FramebufferManagerBase::CopyToXFB(xfbAddr, fbStride, fbHeight, sourceRc, Gamma); } else { - Swap(xfbAddr, fbWidth, fbWidth, fbHeight, sourceRc, Gamma); + // below div two to convert from bytes to pixels - it expects width, not stride + Swap(xfbAddr, fbStride/2, fbStride/2, fbHeight, sourceRc, Gamma); } } @@ -359,15 +362,14 @@ void Renderer::DrawDebugText() case ASPECT_AUTO: ar_text = "Auto"; break; - case ASPECT_FORCE_16_9: - ar_text = "16:9"; - break; - case ASPECT_FORCE_4_3: - ar_text = "4:3"; - break; case ASPECT_STRETCH: ar_text = "Stretch"; break; + case ASPECT_ANALOG: + ar_text = "Force 4:3"; + break; + case ASPECT_ANALOG_WIDE: + ar_text = "Force 16:9"; } const char* const efbcopy_text = g_ActiveConfig.bSkipEFBCopyToRam ? "to Texture" : "to RAM"; @@ -400,7 +402,7 @@ void Renderer::DrawDebugText() } } - final_cyan += Profiler::ToString(); + final_cyan += Common::Profiler::ToString(); if (g_ActiveConfig.bOverlayStats) final_cyan += Statistics::ToString(); @@ -424,31 +426,27 @@ void Renderer::UpdateDrawRectangle(int backbuffer_width, int backbuffer_height) const float WinWidth = FloatGLWidth; const float WinHeight = FloatGLHeight; - // Handle aspect ratio. - // Default to auto. - bool use16_9 = g_aspect_wide; - // Update aspect ratio hack values // Won't take effect until next frame // Don't know if there is a better place for this code so there isn't a 1 frame delay if (g_ActiveConfig.bWidescreenHack) { - float source_aspect = use16_9 ? (16.0f / 9.0f) : (4.0f / 3.0f); + float source_aspect = VideoInterface::GetAspectRatio(g_aspect_wide); float target_aspect; switch (g_ActiveConfig.iAspectRatio) { - case ASPECT_FORCE_16_9: - target_aspect = 16.0f / 9.0f; - break; - case ASPECT_FORCE_4_3: - target_aspect = 4.0f / 3.0f; - break; case ASPECT_STRETCH: target_aspect = WinWidth / WinHeight; break; + case ASPECT_ANALOG: + target_aspect = VideoInterface::GetAspectRatio(false); + break; + case ASPECT_ANALOG_WIDE: + target_aspect = VideoInterface::GetAspectRatio(true); + break; default: - // ASPECT_AUTO == no hacking + // ASPECT_AUTO target_aspect = source_aspect; break; } @@ -475,16 +473,24 @@ void Renderer::UpdateDrawRectangle(int backbuffer_width, int backbuffer_height) } // Check for force-settings and override. - if (g_ActiveConfig.iAspectRatio == ASPECT_FORCE_16_9) - use16_9 = true; - else if (g_ActiveConfig.iAspectRatio == ASPECT_FORCE_4_3) - use16_9 = false; + + // The rendering window aspect ratio as a proportion of the 4:3 or 16:9 ratio + float Ratio; + switch (g_ActiveConfig.iAspectRatio) + { + case ASPECT_ANALOG_WIDE: + Ratio = (WinWidth / WinHeight) / VideoInterface::GetAspectRatio(true); + break; + case ASPECT_ANALOG: + Ratio = (WinWidth / WinHeight) / VideoInterface::GetAspectRatio(false); + break; + default: + Ratio = (WinWidth / WinHeight) / VideoInterface::GetAspectRatio(g_aspect_wide); + break; + } if (g_ActiveConfig.iAspectRatio != ASPECT_STRETCH) { - // The rendering window aspect ratio as a proportion of the 4:3 or 16:9 ratio - float Ratio = (WinWidth / WinHeight) / (!use16_9 ? (4.0f / 3.0f) : (16.0f / 9.0f)); - // Check if height or width is the limiting factor. If ratio > 1 the picture is too wide and have to limit the width. if (Ratio > 1.0f) { // Scale down and center in the X direction. @@ -501,12 +507,27 @@ void Renderer::UpdateDrawRectangle(int backbuffer_width, int backbuffer_height) } // ----------------------------------------------------------------------- - // Crop the picture from 4:3 to 5:4 or from 16:9 to 16:10. + // Crop the picture from Analog to 4:3 or from Analog (Wide) to 16:9. // Output: FloatGLWidth, FloatGLHeight, FloatXOffset, FloatYOffset // ------------------ if (g_ActiveConfig.iAspectRatio != ASPECT_STRETCH && g_ActiveConfig.bCrop) { - float Ratio = !use16_9 ? ((4.0f / 3.0f) / (5.0f / 4.0f)) : (((16.0f / 9.0f) / (16.0f / 10.0f))); + switch (g_ActiveConfig.iAspectRatio) + { + case ASPECT_ANALOG_WIDE: + Ratio = (16.0f / 9.0f) / VideoInterface::GetAspectRatio(true); + break; + case ASPECT_ANALOG: + Ratio = (4.0f / 3.0f) / VideoInterface::GetAspectRatio(false); + break; + default: + Ratio = (!g_aspect_wide ? (4.0f / 3.0f) : (16.0f / 9.0f)) / VideoInterface::GetAspectRatio(g_aspect_wide); + break; + } + if (Ratio <= 1.0f) + { + Ratio = 1.0f / Ratio; + } // The width and height we will add (calculate this before FloatGLWidth and FloatGLHeight is adjusted) float IncreasedWidth = (Ratio - 1.0f) * FloatGLWidth; float IncreasedHeight = (Ratio - 1.0f) * FloatGLHeight; diff --git a/Source/Core/VideoCommon/RenderBase.h b/Source/Core/VideoCommon/RenderBase.h index a17525ca2e..c717483d1e 100644 --- a/Source/Core/VideoCommon/RenderBase.h +++ b/Source/Core/VideoCommon/RenderBase.h @@ -109,7 +109,7 @@ public: virtual void ClearScreen(const EFBRectangle& rc, bool colorEnable, bool alphaEnable, bool zEnable, u32 color, u32 z) = 0; virtual void ReinterpretPixelData(unsigned int convtype) = 0; - static void RenderToXFB(u32 xfbAddr, const EFBRectangle& sourceRc, u32 fbWidth, u32 fbHeight, float Gamma = 1.0f); + static void RenderToXFB(u32 xfbAddr, const EFBRectangle& sourceRc, u32 fbStride, u32 fbHeight, float Gamma = 1.0f); virtual u32 AccessEFB(EFBAccessType type, u32 x, u32 y, u32 poke_data) = 0; virtual void PokeEFB(EFBAccessType type, const std::vector<EfbPokeData>& data); diff --git a/Source/Core/VideoCommon/ShaderGenCommon.h b/Source/Core/VideoCommon/ShaderGenCommon.h index 218d9b98b5..2fd6c51819 100644 --- a/Source/Core/VideoCommon/ShaderGenCommon.h +++ b/Source/Core/VideoCommon/ShaderGenCommon.h @@ -279,6 +279,29 @@ static inline void AssignVSOutputMembers(T& object, const char* a, const char* b } } +// We use the flag "centroid" to fix some MSAA rendering bugs. With MSAA, the +// pixel shader will be executed for each pixel which has at least one passed sample. +// So there may be rendered pixels where the center of the pixel isn't in the primitive. +// As the pixel shader usually renders at the center of the pixel, this position may be +// outside the primitive. This will lead to sampling outside the texture, sign changes, ... +// As a workaround, we interpolate at the centroid of the coveraged pixel, which +// is always inside the primitive. +// Without MSAA, this flag is defined to have no effect. +static inline const char* GetInterpolationQualifier(API_TYPE api_type, bool in = true, bool in_out = false) +{ + if (!g_ActiveConfig.iMultisampleMode) + return ""; + + if (!g_ActiveConfig.bSSAA) + { + if (in_out && api_type == API_OPENGL && !g_ActiveConfig.backend_info.bSupportsBindingLayout) + return in ? "centroid in" : "centroid out"; + return "centroid"; + } + + return "sample"; +} + // Constant variable names #define I_COLORS "color" #define I_KCOLORS "k" diff --git a/Source/Core/VideoCommon/TextureCacheBase.cpp b/Source/Core/VideoCommon/TextureCacheBase.cpp index 0b0588cf6e..1d9b564a8f 100644 --- a/Source/Core/VideoCommon/TextureCacheBase.cpp +++ b/Source/Core/VideoCommon/TextureCacheBase.cpp @@ -10,6 +10,8 @@ #include "Common/StringUtil.h" #include "Core/ConfigManager.h" +#include "Core/FifoPlayer/FifoPlayer.h" +#include "Core/FifoPlayer/FifoRecorder.h" #include "Core/HW/Memmap.h" #include "VideoCommon/Debugger.h" @@ -18,16 +20,17 @@ #include "VideoCommon/RenderBase.h" #include "VideoCommon/Statistics.h" #include "VideoCommon/TextureCacheBase.h" +#include "VideoCommon/VideoCommon.h" #include "VideoCommon/VideoConfig.h" static const u64 TEXHASH_INVALID = 0; -static const int TEXTURE_KILL_THRESHOLD = 60; +static const int TEXTURE_KILL_THRESHOLD = 64; // Sonic the Fighters (inside Sonic Gems Collection) loops a 64 frames animation static const int TEXTURE_POOL_KILL_THRESHOLD = 3; static const int FRAMECOUNT_INVALID = 0; -TextureCache *g_texture_cache; +TextureCache* g_texture_cache; -GC_ALIGNED16(u8 *TextureCache::temp) = nullptr; +alignas(16) u8* TextureCache::temp = nullptr; size_t TextureCache::temp_size; TextureCache::TexCache TextureCache::textures_by_address; @@ -149,12 +152,28 @@ void TextureCache::Cleanup(int _frameCount) if (iter->second->frameCount == FRAMECOUNT_INVALID) { iter->second->frameCount = _frameCount; + ++iter; } - if (_frameCount > TEXTURE_KILL_THRESHOLD + iter->second->frameCount && - // EFB copies living on the host GPU are unrecoverable and thus shouldn't be deleted - !iter->second->IsEfbCopy()) + else if (_frameCount > TEXTURE_KILL_THRESHOLD + iter->second->frameCount) { - iter = FreeTexture(iter); + if (iter->second->IsEfbCopy()) + { + // Only remove EFB copies when they wouldn't be used anymore(changed hash), because EFB copies living on the + // host GPU are unrecoverable. Perform this check only every TEXTURE_KILL_THRESHOLD for performance reasons + if ((_frameCount - iter->second->frameCount) % TEXTURE_KILL_THRESHOLD == 1 && + iter->second->hash != iter->second->CalculateHash()) + { + iter = FreeTexture(iter); + } + else + { + ++iter; + } + } + else + { + iter = FreeTexture(iter); + } } else { @@ -182,24 +201,6 @@ void TextureCache::Cleanup(int _frameCount) } } -void TextureCache::MakeRangeDynamic(u32 start_address, u32 size) -{ - TexCache::iterator - iter = textures_by_address.begin(); - - while (iter != textures_by_address.end()) - { - if (iter->second->OverlapsMemoryRange(start_address, size)) - { - iter = FreeTexture(iter); - } - else - { - ++iter; - } - } -} - bool TextureCache::TCacheEntryBase::OverlapsMemoryRange(u32 range_address, u32 range_size) const { if (addr + size_in_bytes <= range_address) @@ -211,46 +212,98 @@ bool TextureCache::TCacheEntryBase::OverlapsMemoryRange(u32 range_address, u32 r return true; } -void TextureCache::TCacheEntryBase::DoPartialTextureUpdates() +TextureCache::TCacheEntryBase* TextureCache::DoPartialTextureUpdates(TexCache::iterator iter_t) { - const bool isPaletteTexture = (format== GX_TF_C4 || format == GX_TF_C8 || format == GX_TF_C14X2 || format >= 0x10000); + TCacheEntryBase* entry_to_update = iter_t->second; + const bool isPaletteTexture = (entry_to_update->format == GX_TF_C4 + || entry_to_update->format == GX_TF_C8 + || entry_to_update->format == GX_TF_C14X2 + || entry_to_update->format >= 0x10000); // Efb copies and paletted textures are excluded from these updates, until there's an example where a game would // benefit from this. Both would require more work to be done. // TODO: Implement upscaling support for normal textures, and then remove the efb to ram and the scaled efb restrictions - if (!g_ActiveConfig.backend_info.bSupportsCopySubImage || !g_ActiveConfig.bSkipEFBCopyToRam || IsEfbCopy() - || isPaletteTexture || (g_ActiveConfig.bCopyEFBScaled && g_ActiveConfig.iEFBScale != SCALE_1X)) - return; + if (entry_to_update->IsEfbCopy() + || isPaletteTexture) + return entry_to_update; - u32 block_width = TexDecoder_GetBlockWidthInTexels(format); - u32 block_height = TexDecoder_GetBlockHeightInTexels(format); - u32 block_size = block_width * block_height * TexDecoder_GetTexelSizeInNibbles(format) / 2; + u32 block_width = TexDecoder_GetBlockWidthInTexels(entry_to_update->format); + u32 block_height = TexDecoder_GetBlockHeightInTexels(entry_to_update->format); + u32 block_size = block_width * block_height * TexDecoder_GetTexelSizeInNibbles(entry_to_update->format) / 2; - u32 numBlocksX = (native_width + block_width - 1) / block_width; - - TexCache::iterator iter = textures_by_address.lower_bound(addr); - TexCache::iterator iterend = textures_by_address.upper_bound(addr + size_in_bytes); + u32 numBlocksX = (entry_to_update->native_width + block_width - 1) / block_width; + TexCache::iterator iter = textures_by_address.lower_bound(entry_to_update->addr); + TexCache::iterator iterend = textures_by_address.upper_bound(entry_to_update->addr + entry_to_update->size_in_bytes); + bool entry_need_scaling = true; while (iter != iterend) { TCacheEntryBase* entry = iter->second; - if (entry->IsEfbCopy() && addr <= entry->addr && entry->addr + entry->size_in_bytes <= addr + size_in_bytes - && entry->frameCount == FRAMECOUNT_INVALID && entry->copyMipMapStrideChannels * 32 == numBlocksX * block_size) + if (entry != entry_to_update + && entry->IsEfbCopy() + && entry_to_update->addr <= entry->addr + && entry->addr + entry->size_in_bytes <= entry_to_update->addr + entry_to_update->size_in_bytes + && entry->frameCount == FRAMECOUNT_INVALID + && entry->memory_stride == numBlocksX * block_size) { - u32 block_offset = (entry->addr - addr) / block_size; + u32 block_offset = (entry->addr - entry_to_update->addr) / block_size; u32 block_x = block_offset % numBlocksX; u32 block_y = block_offset / numBlocksX; u32 x = block_x * block_width; u32 y = block_y * block_height; - - DoPartialTextureUpdate(entry, x, y); - + MathUtil::Rectangle<int> srcrect, dstrect; + srcrect.left = 0; + srcrect.top = 0; + dstrect.left = 0; + dstrect.top = 0; + if (entry_need_scaling) + { + entry_need_scaling = false; + u32 w = entry_to_update->native_width * entry->config.width / entry->native_width; + u32 h = entry_to_update->native_height * entry->config.height / entry->native_height; + u32 max = g_renderer->GetMaxTextureSize(); + if (max < w || max < h) + { + iter++; + continue; + } + if (entry_to_update->config.width != w || entry_to_update->config.height != h) + { + TextureCache::TCacheEntryConfig newconfig; + newconfig.width = w; + newconfig.height = h; + newconfig.rendertarget = true; + TCacheEntryBase* newentry = AllocateTexture(newconfig); + newentry->SetGeneralParameters(entry_to_update->addr, entry_to_update->size_in_bytes, entry_to_update->format); + newentry->SetDimensions(entry_to_update->native_width, entry_to_update->native_height, 1); + newentry->SetHashes(entry_to_update->base_hash, entry_to_update->hash); + newentry->frameCount = frameCount; + newentry->is_efb_copy = false; + srcrect.right = entry_to_update->config.width; + srcrect.bottom = entry_to_update->config.height; + dstrect.right = w; + dstrect.bottom = h; + newentry->CopyRectangleFromTexture(entry_to_update, srcrect, dstrect); + entry_to_update = newentry; + u64 key = iter_t->first; + iter_t = FreeTexture(iter_t); + textures_by_address.emplace(key, entry_to_update); + } + } + srcrect.right = entry->config.width; + srcrect.bottom = entry->config.height; + dstrect.left = x * entry_to_update->config.width / entry_to_update->native_width; + dstrect.top = y * entry_to_update->config.height / entry_to_update->native_height; + dstrect.right = (x + entry->native_width) * entry_to_update->config.width / entry_to_update->native_width; + dstrect.bottom = (y + entry->native_height) * entry_to_update->config.height / entry_to_update->native_height; + entry_to_update->CopyRectangleFromTexture(entry, srcrect, dstrect); // Mark the texture update as used, so it isn't applied more than once entry->frameCount = frameCount; } ++iter; } + return entry_to_update; } void TextureCache::DumpTexture(TCacheEntryBase* entry, std::string basename, unsigned int level) @@ -320,16 +373,16 @@ TextureCache::TCacheEntryBase* TextureCache::Load(const u32 stage) return nullptr; // TexelSizeInNibbles(format) * width * height / 16; - const unsigned int bsw = TexDecoder_GetBlockWidthInTexels(texformat) - 1; - const unsigned int bsh = TexDecoder_GetBlockHeightInTexels(texformat) - 1; + const unsigned int bsw = TexDecoder_GetBlockWidthInTexels(texformat); + const unsigned int bsh = TexDecoder_GetBlockHeightInTexels(texformat); - unsigned int expandedWidth = (width + bsw) & (~bsw); - unsigned int expandedHeight = (height + bsh) & (~bsh); + unsigned int expandedWidth = ROUND_UP(width, bsw); + unsigned int expandedHeight = ROUND_UP(height, bsh); const unsigned int nativeW = width; const unsigned int nativeH = height; // Hash assigned to texcache entry (also used to generate filenames used for texture dumping and custom texture lookup) - u64 tex_hash = TEXHASH_INVALID; + u64 base_hash = TEXHASH_INVALID; u64 full_hash = TEXHASH_INVALID; u32 full_format = texformat; @@ -344,6 +397,25 @@ TextureCache::TCacheEntryBase* TextureCache::Load(const u32 stage) full_format = texformat | (tlutfmt << 16); const u32 texture_size = TexDecoder_GetTextureSizeInBytes(expandedWidth, expandedHeight, texformat); + u32 additional_mips_size = 0; // not including level 0, which is texture_size + + // GPUs don't like when the specified mipmap count would require more than one 1x1-sized LOD in the mipmap chain + // e.g. 64x64 with 7 LODs would have the mipmap chain 64x64,32x32,16x16,8x8,4x4,2x2,1x1,0x0, so we limit the mipmap count to 6 there + tex_levels = std::min<u32>(IntLog2(std::max(width, height)) + 1, tex_levels); + + for (u32 level = 1; level != tex_levels; ++level) + { + // We still need to calculate the original size of the mips + const u32 expanded_mip_width = ROUND_UP(CalculateLevelSize(width, level), bsw); + const u32 expanded_mip_height = ROUND_UP(CalculateLevelSize(height, level), bsh); + + additional_mips_size += TexDecoder_GetTextureSizeInBytes(expanded_mip_width, expanded_mip_height, texformat); + } + + // If we are recording a FifoLog, keep track of what memory we read. + // FifiRecorder does it's own memory modification tracking independant of the texture hashing below. + if (g_bRecordFifoData && !from_tmem) + FifoRecorder::GetInstance().UseMemory(address, texture_size + additional_mips_size, MemoryUpdate::TEXTURE_MAP); const u8* src_data; if (from_tmem) @@ -352,22 +424,18 @@ TextureCache::TCacheEntryBase* TextureCache::Load(const u32 stage) src_data = Memory::GetPointer(address); // TODO: This doesn't hash GB tiles for preloaded RGBA8 textures (instead, it's hashing more data from the low tmem bank than it should) - tex_hash = GetHash64(src_data, texture_size, g_ActiveConfig.iSafeTextureCache_ColorSamples); + base_hash = GetHash64(src_data, texture_size, g_ActiveConfig.iSafeTextureCache_ColorSamples); u32 palette_size = 0; if (isPaletteTexture) { palette_size = TexDecoder_GetPaletteSize(texformat); - full_hash = tex_hash ^ GetHash64(&texMem[tlutaddr], palette_size, g_ActiveConfig.iSafeTextureCache_ColorSamples); + full_hash = base_hash ^ GetHash64(&texMem[tlutaddr], palette_size, g_ActiveConfig.iSafeTextureCache_ColorSamples); } else { - full_hash = tex_hash; + full_hash = base_hash; } - // GPUs don't like when the specified mipmap count would require more than one 1x1-sized LOD in the mipmap chain - // e.g. 64x64 with 7 LODs would have the mipmap chain 64x64,32x32,16x16,8x8,4x4,2x2,1x1,0x0, so we limit the mipmap count to 6 there - tex_levels = std::min<u32>(IntLog2(std::max(width, height)) + 1, tex_levels); - // Search the texture cache for textures by address // // Find all texture cache entries for the current texture address, and decide whether to use one of @@ -405,11 +473,10 @@ TextureCache::TCacheEntryBase* TextureCache::Load(const u32 stage) TCacheEntryBase* entry = iter->second; if (entry->IsEfbCopy()) { - // EFB copies have slightly different rules: the hash doesn't need to match - // in EFB2Tex mode, and EFB copy formats have different meanings from texture - // formats. - if (g_ActiveConfig.bSkipEFBCopyToRam || - (tex_hash == entry->hash && (!isPaletteTexture || g_Config.backend_info.bSupportsPaletteConversion))) + // EFB copies have slightly different rules as EFB copy formats have different + // meanings from texture formats. + if ((base_hash == entry->hash && (!isPaletteTexture || g_Config.backend_info.bSupportsPaletteConversion)) || + IsPlayingBackFifologWithBrokenEFBCopies) { // TODO: We should check format/width/height/levels for EFB copies. Checking // format is complicated because EFB copy formats don't exactly match @@ -440,14 +507,18 @@ TextureCache::TCacheEntryBase* TextureCache::Load(const u32 stage) if (entry->hash == full_hash && entry->format == full_format && entry->native_levels >= tex_levels && entry->native_width == nativeW && entry->native_height == nativeH) { - entry->DoPartialTextureUpdates(); + entry = DoPartialTextureUpdates(iter); return ReturnEntry(stage, entry); } } - // Find the entry which hasn't been used for the longest time - if (entry->frameCount != FRAMECOUNT_INVALID && entry->frameCount < temp_frameCount) + // Find the texture which hasn't been used for the longest time. Count paletted + // textures as the same texture here, when the texture itself is the same. This + // improves the performance a lot in some games that use paletted textures. + // Example: Sonic the Fighters (inside Sonic Gems Collection) + if (entry->frameCount != FRAMECOUNT_INVALID && entry->frameCount < temp_frameCount && + !(isPaletteTexture && entry->base_hash == base_hash)) { temp_frameCount = entry->frameCount; oldest_entry = iter; @@ -469,12 +540,12 @@ TextureCache::TCacheEntryBase* TextureCache::Load(const u32 stage) decoded_entry->SetGeneralParameters(address, texture_size, full_format); decoded_entry->SetDimensions(entry->native_width, entry->native_height, 1); - decoded_entry->SetHashes(full_hash); + decoded_entry->SetHashes(base_hash, full_hash); decoded_entry->frameCount = FRAMECOUNT_INVALID; decoded_entry->is_efb_copy = false; g_texture_cache->ConvertTexture(decoded_entry, entry, &texMem[tlutaddr], (TlutFormat)tlutfmt); - textures_by_address.insert(TexCache::value_type((u64)address, decoded_entry)); + textures_by_address.emplace((u64)address, decoded_entry); return ReturnEntry(stage, decoded_entry); } @@ -494,7 +565,7 @@ TextureCache::TCacheEntryBase* TextureCache::Load(const u32 stage) if (entry->format == full_format && entry->native_levels >= tex_levels && entry->native_width == nativeW && entry->native_height == nativeH) { - entry->DoPartialTextureUpdates(); + entry = DoPartialTextureUpdates(iter); return ReturnEntry(stage, entry); } @@ -539,7 +610,7 @@ TextureCache::TCacheEntryBase* TextureCache::Load(const u32 stage) if (!(texformat == GX_TF_RGBA8 && from_tmem)) { const u8* tlut = &texMem[tlutaddr]; - TexDecoder_Decode(temp, src_data, expandedWidth, expandedHeight, texformat, tlut, (TlutFormat) tlutfmt); + TexDecoder_Decode(temp, src_data, expandedWidth, expandedHeight, texformat, tlut, (TlutFormat)tlutfmt); } else { @@ -560,16 +631,16 @@ TextureCache::TCacheEntryBase* TextureCache::Load(const u32 stage) TCacheEntryBase* entry = AllocateTexture(config); GFX_DEBUGGER_PAUSE_AT(NEXT_NEW_TEXTURE, true); - textures_by_address.insert(TexCache::value_type((u64)address, entry)); + iter = textures_by_address.emplace((u64)address, entry); if (g_ActiveConfig.iSafeTextureCache_ColorSamples == 0 || std::max(texture_size, palette_size) <= (u32)g_ActiveConfig.iSafeTextureCache_ColorSamples * 8) { - entry->textures_by_hash_iter = textures_by_hash.insert(TexCache::value_type(full_hash, entry)); + entry->textures_by_hash_iter = textures_by_hash.emplace(full_hash, entry); } entry->SetGeneralParameters(address, texture_size, full_format); entry->SetDimensions(nativeW, nativeH, tex_levels); - entry->hash = full_hash; + entry->SetHashes(base_hash, full_hash); entry->is_efb_copy = false; entry->is_custom_tex = hires_tex != nullptr; @@ -616,8 +687,8 @@ TextureCache::TCacheEntryBase* TextureCache::Load(const u32 stage) { const u32 mip_width = CalculateLevelSize(width, level); const u32 mip_height = CalculateLevelSize(height, level); - const u32 expanded_mip_width = (mip_width + bsw) & (~bsw); - const u32 expanded_mip_height = (mip_height + bsh) & (~bsh); + const u32 expanded_mip_width = ROUND_UP(mip_width, bsw); + const u32 expanded_mip_height = ROUND_UP(mip_height, bsh); const u8*& mip_src_data = from_tmem ? ((level % 2) ? ptr_odd : ptr_even) @@ -636,12 +707,12 @@ TextureCache::TCacheEntryBase* TextureCache::Load(const u32 stage) INCSTAT(stats.numTexturesUploaded); SETSTAT(stats.numTexturesAlive, textures_by_address.size()); - entry->DoPartialTextureUpdates(); + entry = DoPartialTextureUpdates(iter); return ReturnEntry(stage, entry); } -void TextureCache::CopyRenderTargetToTexture(u32 dstAddr, unsigned int dstFormat, PEControl::PixelFormat srcFormat, +void TextureCache::CopyRenderTargetToTexture(u32 dstAddr, unsigned int dstFormat, u32 dstStride, PEControl::PixelFormat srcFormat, const EFBRectangle& srcRect, bool isIntensity, bool scaleByHalf) { // Emulation methods: @@ -686,7 +757,7 @@ void TextureCache::CopyRenderTargetToTexture(u32 dstAddr, unsigned int dstFormat // // For historical reasons, Dolphin doesn't actually implement "pure" EFB to RAM emulation, but only EFB to texture and hybrid EFB copies. - float colmat[28] = {0}; + float colmat[28] = { 0 }; float *const fConstAdd = colmat + 16; float *const ColorMask = colmat + 20; ColorMask[0] = ColorMask[1] = ColorMask[2] = ColorMask[3] = 255.0f; @@ -701,9 +772,11 @@ void TextureCache::CopyRenderTargetToTexture(u32 dstAddr, unsigned int dstFormat case 0: // Z4 colmat[3] = colmat[7] = colmat[11] = colmat[15] = 1.0f; cbufid = 0; + dstFormat |= _GX_TF_CTF; break; + case 8: // Z8H + dstFormat |= _GX_TF_CTF; case 1: // Z8 - case 8: // Z8 colmat[0] = colmat[4] = colmat[8] = colmat[12] = 1.0f; cbufid = 1; break; @@ -716,6 +789,7 @@ void TextureCache::CopyRenderTargetToTexture(u32 dstAddr, unsigned int dstFormat case 11: // Z16 (reverse order) colmat[0] = colmat[4] = colmat[8] = colmat[13] = 1.0f; cbufid = 3; + dstFormat |= _GX_TF_CTF; break; case 6: // Z24X8 @@ -726,11 +800,13 @@ void TextureCache::CopyRenderTargetToTexture(u32 dstAddr, unsigned int dstFormat case 9: // Z8M colmat[1] = colmat[5] = colmat[9] = colmat[13] = 1.0f; cbufid = 5; + dstFormat |= _GX_TF_CTF; break; case 10: // Z8L colmat[2] = colmat[6] = colmat[10] = colmat[14] = 1.0f; cbufid = 6; + dstFormat |= _GX_TF_CTF; break; case 12: // Z16L - copy lower 16 depth bits @@ -738,6 +814,7 @@ void TextureCache::CopyRenderTargetToTexture(u32 dstAddr, unsigned int dstFormat // Used e.g. in Zelda: Skyward Sword colmat[1] = colmat[5] = colmat[9] = colmat[14] = 1.0f; cbufid = 7; + dstFormat |= _GX_TF_CTF; break; default: @@ -746,6 +823,8 @@ void TextureCache::CopyRenderTargetToTexture(u32 dstAddr, unsigned int dstFormat cbufid = 8; break; } + + dstFormat |= _GX_TF_ZTF; } else if (isIntensity) { @@ -810,11 +889,13 @@ void TextureCache::CopyRenderTargetToTexture(u32 dstAddr, unsigned int dstFormat ColorMask[0] = 15.0f; ColorMask[4] = 1.0f / 15.0f; cbufid = 14; + dstFormat |= _GX_TF_CTF; break; case 1: // R8 case 8: // R8 colmat[0] = colmat[4] = colmat[8] = colmat[12] = 1; cbufid = 15; + dstFormat |= _GX_TF_CTF; break; case 2: // RA4 @@ -829,6 +910,7 @@ void TextureCache::CopyRenderTargetToTexture(u32 dstAddr, unsigned int dstFormat fConstAdd[3] = 1.0f; cbufid = 17; } + dstFormat |= _GX_TF_CTF; break; case 3: // RA8 colmat[0] = colmat[4] = colmat[8] = colmat[15] = 1.0f; @@ -840,6 +922,7 @@ void TextureCache::CopyRenderTargetToTexture(u32 dstAddr, unsigned int dstFormat fConstAdd[3] = 1.0f; cbufid = 19; } + dstFormat |= _GX_TF_CTF; break; case 7: // A8 @@ -855,25 +938,30 @@ void TextureCache::CopyRenderTargetToTexture(u32 dstAddr, unsigned int dstFormat fConstAdd[3] = 1.0f; cbufid = 21; } + dstFormat |= _GX_TF_CTF; break; case 9: // G8 colmat[1] = colmat[5] = colmat[9] = colmat[13] = 1.0f; cbufid = 22; + dstFormat |= _GX_TF_CTF; break; case 10: // B8 colmat[2] = colmat[6] = colmat[10] = colmat[14] = 1.0f; cbufid = 23; + dstFormat |= _GX_TF_CTF; break; case 11: // RG8 colmat[0] = colmat[4] = colmat[8] = colmat[13] = 1.0f; cbufid = 24; + dstFormat |= _GX_TF_CTF; break; case 12: // GB8 colmat[1] = colmat[5] = colmat[9] = colmat[14] = 1.0f; cbufid = 25; + dstFormat |= _GX_TF_CTF; break; case 4: // RGB565 @@ -921,6 +1009,13 @@ void TextureCache::CopyRenderTargetToTexture(u32 dstAddr, unsigned int dstFormat } } + u8* dst = Memory::GetPointer(dstAddr); + if (dst == nullptr) + { + ERROR_LOG(VIDEO, "Trying to copy from EFB to invalid address 0x%8x", dstAddr); + return; + } + const unsigned int tex_w = scaleByHalf ? srcRect.GetWidth() / 2 : srcRect.GetWidth(); const unsigned int tex_h = scaleByHalf ? srcRect.GetHeight() / 2 : srcRect.GetHeight(); @@ -928,11 +1023,13 @@ void TextureCache::CopyRenderTargetToTexture(u32 dstAddr, unsigned int dstFormat unsigned int scaled_tex_h = g_ActiveConfig.bCopyEFBScaled ? Renderer::EFBToScaledY(tex_h) : tex_h; // remove all texture cache entries at dstAddr - std::pair<TexCache::iterator, TexCache::iterator> iter_range = textures_by_address.equal_range((u64)dstAddr); - TexCache::iterator iter = iter_range.first; - while (iter != iter_range.second) { - iter = FreeTexture(iter); + std::pair<TexCache::iterator, TexCache::iterator> iter_range = textures_by_address.equal_range((u64)dstAddr); + TexCache::iterator iter = iter_range.first; + while (iter != iter_range.second) + { + iter = FreeTexture(iter); + } } // create the texture @@ -944,17 +1041,32 @@ void TextureCache::CopyRenderTargetToTexture(u32 dstAddr, unsigned int dstFormat TCacheEntryBase* entry = AllocateTexture(config); - // TODO: Using the wrong dstFormat, dumb... entry->SetGeneralParameters(dstAddr, 0, dstFormat); entry->SetDimensions(tex_w, tex_h, 1); - entry->SetHashes(TEXHASH_INVALID); entry->frameCount = FRAMECOUNT_INVALID; - entry->is_efb_copy = true; + entry->SetEfbCopy(dstStride); entry->is_custom_tex = false; - entry->copyMipMapStrideChannels = bpmem.copyMipMapStrideChannels; - entry->FromRenderTarget(dstAddr, dstFormat, srcFormat, srcRect, isIntensity, scaleByHalf, cbufid, colmat); + entry->FromRenderTarget(dst, dstFormat, dstStride, srcFormat, srcRect, isIntensity, scaleByHalf, cbufid, colmat); + + u64 hash = entry->CalculateHash(); + entry->SetHashes(hash, hash); + + // Invalidate all textures that overlap the range of our efb copy. + // Unless our efb copy has a weird stride, then we want avoid invalidating textures which + // we might be able to do a partial texture update on. + if (entry->memory_stride == entry->CacheLinesPerRow() * 32) + { + TexCache::iterator iter = textures_by_address.begin(); + while (iter != textures_by_address.end()) + { + if (iter->second->OverlapsMemoryRange(dstAddr, entry->size_in_bytes)) + iter = FreeTexture(iter); + else + ++iter; + } + } if (g_ActiveConfig.bDumpEFBTarget) { @@ -963,7 +1075,18 @@ void TextureCache::CopyRenderTargetToTexture(u32 dstAddr, unsigned int dstFormat count++), 0); } - textures_by_address.insert(TexCache::value_type((u64)dstAddr, entry)); + if (g_bRecordFifoData) + { + // Mark the memory behind this efb copy as dynamicly generated for the Fifo log + u32 address = dstAddr; + for (u32 i = 0; i < entry->NumBlocksY(); i++) + { + FifoRecorder::GetInstance().UseMemory(address, entry->CacheLinesPerRow() * 32, MemoryUpdate::TEXTURE_MAP, true); + address += entry->memory_stride; + } + } + + textures_by_address.emplace((u64)dstAddr, entry); } TextureCache::TCacheEntryBase* TextureCache::AllocateTexture(const TCacheEntryConfig& config) @@ -996,7 +1119,79 @@ TextureCache::TexCache::iterator TextureCache::FreeTexture(TexCache::iterator it } entry->frameCount = FRAMECOUNT_INVALID; - texture_pool.insert(TexPool::value_type(entry->config, entry)); + texture_pool.emplace(entry->config, entry); return textures_by_address.erase(iter); } + +u32 TextureCache::TCacheEntryBase::CacheLinesPerRow() const +{ + u32 blockW = TexDecoder_GetBlockWidthInTexels(format); + // Round up source height to multiple of block size + u32 actualWidth = ROUND_UP(native_width, blockW); + + u32 numBlocksX = actualWidth / blockW; + + // RGBA takes two cache lines per block; all others take one + if (format == GX_TF_RGBA8) + numBlocksX = numBlocksX * 2; + return numBlocksX; +} + +u32 TextureCache::TCacheEntryBase::NumBlocksY() const +{ + u32 blockH = TexDecoder_GetBlockHeightInTexels(format); + // Round up source height to multiple of block size + u32 actualHeight = ROUND_UP(native_height, blockH); + + return actualHeight / blockH; +} + +void TextureCache::TCacheEntryBase::SetEfbCopy(u32 stride) +{ + is_efb_copy = true; + memory_stride = stride; + + _assert_msg_(VIDEO, memory_stride >= CacheLinesPerRow(), "Memory stride is too small"); + + size_in_bytes = memory_stride * NumBlocksY(); +} + +// Fill gamecube memory backing this texture with zeros. +void TextureCache::TCacheEntryBase::Zero(u8* ptr) +{ + for (u32 i = 0; i < NumBlocksY(); i++) + { + memset(ptr, 0, CacheLinesPerRow() * 32); + ptr += memory_stride; + } +} + +u64 TextureCache::TCacheEntryBase::CalculateHash() const +{ + u8* ptr = Memory::GetPointer(addr); + if (memory_stride == CacheLinesPerRow() * 32) + { + return GetHash64(ptr, size_in_bytes, g_ActiveConfig.iSafeTextureCache_ColorSamples); + } + else + { + u32 blocks = NumBlocksY(); + u64 temp_hash = size_in_bytes; + + u32 samples_per_row = 0; + if (g_ActiveConfig.iSafeTextureCache_ColorSamples != 0) + { + // Hash at least 4 samples per row to avoid hashing in a bad pattern, like just on the left side of the efb copy + samples_per_row = std::max(g_ActiveConfig.iSafeTextureCache_ColorSamples / blocks, 4u); + } + + for (u32 i = 0; i < blocks; i++) + { + // Multiply by a prime number to mix the hash up a bit. This prevents identical blocks from canceling each other out + temp_hash = (temp_hash * 397) ^ GetHash64(ptr, CacheLinesPerRow() * 32, samples_per_row); + ptr += memory_stride; + } + return temp_hash; + } +} diff --git a/Source/Core/VideoCommon/TextureCacheBase.h b/Source/Core/VideoCommon/TextureCacheBase.h index 5d4f9204fc..4fbaed8e15 100644 --- a/Source/Core/VideoCommon/TextureCacheBase.h +++ b/Source/Core/VideoCommon/TextureCacheBase.h @@ -49,11 +49,12 @@ public: // common members u32 addr; u32 size_in_bytes; - u64 hash; + u64 base_hash; + u64 hash; // for paletted textures, hash = base_hash ^ palette_hash u32 format; bool is_efb_copy; bool is_custom_tex; - u32 copyMipMapStrideChannels; + u32 memory_stride; unsigned int native_width, native_height; // Texture dimensions from the GameCube's point of view unsigned int native_levels; @@ -76,33 +77,45 @@ public: native_width = _native_width; native_height = _native_height; native_levels = _native_levels; + memory_stride = _native_width; } - void SetHashes(u64 _hash) + void SetHashes(u64 _base_hash, u64 _hash) { + base_hash = _base_hash; hash = _hash; } + void SetEfbCopy(u32 stride); + TCacheEntryBase(const TCacheEntryConfig& c) : config(c) {} virtual ~TCacheEntryBase(); virtual void Bind(unsigned int stage) = 0; virtual bool Save(const std::string& filename, unsigned int level) = 0; - virtual void DoPartialTextureUpdate(TCacheEntryBase* entry, u32 x, u32 y) = 0; + virtual void CopyRectangleFromTexture( + const TCacheEntryBase* source, + const MathUtil::Rectangle<int> &srcrect, + const MathUtil::Rectangle<int> &dstrect) = 0; virtual void Load(unsigned int width, unsigned int height, unsigned int expanded_width, unsigned int level) = 0; - virtual void FromRenderTarget(u32 dstAddr, unsigned int dstFormat, + virtual void FromRenderTarget(u8* dst, unsigned int dstFormat, u32 dstStride, PEControl::PixelFormat srcFormat, const EFBRectangle& srcRect, bool isIntensity, bool scaleByHalf, unsigned int cbufid, const float *colmat) = 0; bool OverlapsMemoryRange(u32 range_address, u32 range_size) const; - void DoPartialTextureUpdates(); - bool IsEfbCopy() const { return is_efb_copy; } + + u32 NumBlocksY() const; + u32 CacheLinesPerRow() const; + + void Zero(u8* ptr); + + u64 CalculateHash() const; }; virtual ~TextureCache(); // needs virtual for DX11 dtor @@ -114,7 +127,6 @@ public: static void Cleanup(int _frameCount); static void Invalidate(); - static void MakeRangeDynamic(u32 start_address, u32 size); virtual TCacheEntryBase* CreateTexture(const TCacheEntryConfig& config) = 0; @@ -124,8 +136,8 @@ public: static TCacheEntryBase* Load(const u32 stage); static void UnbindTextures(); static void BindTextures(); - static void CopyRenderTargetToTexture(u32 dstAddr, unsigned int dstFormat, PEControl::PixelFormat srcFormat, - const EFBRectangle& srcRect, bool isIntensity, bool scaleByHalf); + static void CopyRenderTargetToTexture(u32 dstAddr, unsigned int dstFormat, u32 dstStride, + PEControl::PixelFormat srcFormat, const EFBRectangle& srcRect, bool isIntensity, bool scaleByHalf); static void RequestInvalidateTextureCache(); @@ -134,13 +146,13 @@ public: protected: TextureCache(); - static GC_ALIGNED16(u8 *temp); + alignas(16) static u8* temp; static size_t temp_size; private: typedef std::multimap<u64, TCacheEntryBase*> TexCache; typedef std::unordered_multimap<TCacheEntryConfig, TCacheEntryBase*, TCacheEntryConfig::Hasher> TexPool; - + static TCacheEntryBase* DoPartialTextureUpdates(TexCache::iterator iter); static void DumpTexture(TCacheEntryBase* entry, std::string basename, unsigned int level); static void CheckTempSize(size_t required_size); diff --git a/Source/Core/VideoCommon/TextureConversionShader.cpp b/Source/Core/VideoCommon/TextureConversionShader.cpp index f93d465a16..033a56b4aa 100644 --- a/Source/Core/VideoCommon/TextureConversionShader.cpp +++ b/Source/Core/VideoCommon/TextureConversionShader.cpp @@ -148,7 +148,7 @@ static void WriteToBitDepth(char*& p, u8 depth, const char* src, const char* des WRITE(p, " %s = floor(%s * 255.0 / exp2(8.0 - %d.0));\n", dest, src, depth); } -static void WriteEncoderEnd(char*& p, API_TYPE ApiType) +static void WriteEncoderEnd(char*& p) { WRITE(p, "}\n"); IntensityConstantAdded = false; @@ -173,7 +173,7 @@ static void WriteI8Encoder(char*& p, API_TYPE ApiType) WRITE(p, " ocol0.rgba += IntensityConst.aaaa;\n"); // see WriteColorToIntensity - WriteEncoderEnd(p, ApiType); + WriteEncoderEnd(p); } static void WriteI4Encoder(char*& p, API_TYPE ApiType) @@ -214,7 +214,7 @@ static void WriteI4Encoder(char*& p, API_TYPE ApiType) WriteToBitDepth(p, 4, "color1", "color1"); WRITE(p, " ocol0 = (color0 * 16.0 + color1) / 255.0;\n"); - WriteEncoderEnd(p, ApiType); + WriteEncoderEnd(p); } static void WriteIA8Encoder(char*& p,API_TYPE ApiType) @@ -232,7 +232,7 @@ static void WriteIA8Encoder(char*& p,API_TYPE ApiType) WRITE(p, " ocol0.ga += IntensityConst.aa;\n"); - WriteEncoderEnd(p, ApiType); + WriteEncoderEnd(p); } static void WriteIA4Encoder(char*& p,API_TYPE ApiType) @@ -264,7 +264,7 @@ static void WriteIA4Encoder(char*& p,API_TYPE ApiType) WriteToBitDepth(p, 4, "color1", "color1"); WRITE(p, " ocol0 = (color0 * 16.0 + color1) / 255.0;\n"); - WriteEncoderEnd(p, ApiType); + WriteEncoderEnd(p); } static void WriteRGB565Encoder(char*& p,API_TYPE ApiType) @@ -287,7 +287,7 @@ static void WriteRGB565Encoder(char*& p,API_TYPE ApiType) WRITE(p, " ocol0.ga = ocol0.ga + gLower * 32.0;\n"); WRITE(p, " ocol0 = ocol0 / 255.0;\n"); - WriteEncoderEnd(p, ApiType); + WriteEncoderEnd(p); } static void WriteRGB5A3Encoder(char*& p,API_TYPE ApiType) @@ -353,7 +353,7 @@ static void WriteRGB5A3Encoder(char*& p,API_TYPE ApiType) WRITE(p, "}\n"); WRITE(p, " ocol0 = ocol0 / 255.0;\n"); - WriteEncoderEnd(p, ApiType); + WriteEncoderEnd(p); } static void WriteRGBA8Encoder(char*& p,API_TYPE ApiType) @@ -378,7 +378,7 @@ static void WriteRGBA8Encoder(char*& p,API_TYPE ApiType) WRITE(p, " ocol0 = first ? color0 : color1;\n"); - WriteEncoderEnd(p, ApiType); + WriteEncoderEnd(p); } static void WriteC4Encoder(char*& p, const char* comp,API_TYPE ApiType) @@ -400,7 +400,7 @@ static void WriteC4Encoder(char*& p, const char* comp,API_TYPE ApiType) WriteToBitDepth(p, 4, "color1", "color1"); WRITE(p, " ocol0 = (color0 * 16.0 + color1) / 255.0;\n"); - WriteEncoderEnd(p, ApiType); + WriteEncoderEnd(p); } static void WriteC8Encoder(char*& p, const char* comp,API_TYPE ApiType) @@ -412,7 +412,7 @@ static void WriteC8Encoder(char*& p, const char* comp,API_TYPE ApiType) WriteSampleColor(p, comp, "ocol0.r", 2, ApiType); WriteSampleColor(p, comp, "ocol0.a", 3, ApiType); - WriteEncoderEnd(p, ApiType); + WriteEncoderEnd(p); } static void WriteCC4Encoder(char*& p, const char* comp,API_TYPE ApiType) @@ -442,7 +442,7 @@ static void WriteCC4Encoder(char*& p, const char* comp,API_TYPE ApiType) WriteToBitDepth(p, 4, "color1", "color1"); WRITE(p, " ocol0 = (color0 * 16.0 + color1) / 255.0;\n"); - WriteEncoderEnd(p, ApiType); + WriteEncoderEnd(p); } static void WriteCC8Encoder(char*& p, const char* comp, API_TYPE ApiType) @@ -452,7 +452,7 @@ static void WriteCC8Encoder(char*& p, const char* comp, API_TYPE ApiType) WriteSampleColor(p, comp, "ocol0.bg", 0, ApiType); WriteSampleColor(p, comp, "ocol0.ra", 1, ApiType); - WriteEncoderEnd(p, ApiType); + WriteEncoderEnd(p); } static void WriteZ8Encoder(char*& p, const char* multiplier,API_TYPE ApiType) @@ -477,7 +477,7 @@ static void WriteZ8Encoder(char*& p, const char* multiplier,API_TYPE ApiType) if (ApiType == API_D3D) WRITE(p, "depth = 1.0f - depth;\n"); WRITE(p, "ocol0.a = frac(depth * %s);\n", multiplier); - WriteEncoderEnd(p, ApiType); + WriteEncoderEnd(p); } static void WriteZ16Encoder(char*& p,API_TYPE ApiType) @@ -492,7 +492,7 @@ static void WriteZ16Encoder(char*& p,API_TYPE ApiType) WriteSampleColor(p, "r", "depth", 0, ApiType); if (ApiType == API_D3D) WRITE(p, "depth = 1.0f - depth;\n"); - WRITE(p, " depth = clamp(depth * 16777216.0, 0.0, float(0xFFFFFF));\n"); + WRITE(p, " depth *= 16777216.0;\n"); WRITE(p, " expanded.r = floor(depth / (256.0 * 256.0));\n"); WRITE(p, " depth -= expanded.r * 256.0 * 256.0;\n"); WRITE(p, " expanded.g = floor(depth / 256.0);\n"); @@ -503,7 +503,7 @@ static void WriteZ16Encoder(char*& p,API_TYPE ApiType) WriteSampleColor(p, "r", "depth", 1, ApiType); if (ApiType == API_D3D) WRITE(p, "depth = 1.0f - depth;\n"); - WRITE(p, " depth = clamp(depth * 16777216.0, 0.0, float(0xFFFFFF));\n"); + WRITE(p, " depth *= 16777216.0;\n"); WRITE(p, " expanded.r = floor(depth / (256.0 * 256.0));\n"); WRITE(p, " depth -= expanded.r * 256.0 * 256.0;\n"); WRITE(p, " expanded.g = floor(depth / 256.0);\n"); @@ -511,7 +511,7 @@ static void WriteZ16Encoder(char*& p,API_TYPE ApiType) WRITE(p, " ocol0.r = expanded.g / 255.0;\n"); WRITE(p, " ocol0.a = expanded.r / 255.0;\n"); - WriteEncoderEnd(p, ApiType); + WriteEncoderEnd(p); } static void WriteZ16LEncoder(char*& p,API_TYPE ApiType) @@ -526,7 +526,7 @@ static void WriteZ16LEncoder(char*& p,API_TYPE ApiType) WriteSampleColor(p, "r", "depth", 0, ApiType); if (ApiType == API_D3D) WRITE(p, "depth = 1.0f - depth;\n"); - WRITE(p, " depth = clamp(depth * 16777216.0, 0.0, float(0xFFFFFF));\n"); + WRITE(p, " depth *= 16777216.0;\n"); WRITE(p, " expanded.r = floor(depth / (256.0 * 256.0));\n"); WRITE(p, " depth -= expanded.r * 256.0 * 256.0;\n"); WRITE(p, " expanded.g = floor(depth / 256.0);\n"); @@ -539,7 +539,7 @@ static void WriteZ16LEncoder(char*& p,API_TYPE ApiType) WriteSampleColor(p, "r", "depth", 1, ApiType); if (ApiType == API_D3D) WRITE(p, "depth = 1.0f - depth;\n"); - WRITE(p, " depth = clamp(depth * 16777216.0, 0.0, float(0xFFFFFF));\n"); + WRITE(p, " depth *= 16777216.0;\n"); WRITE(p, " expanded.r = floor(depth / (256.0 * 256.0));\n"); WRITE(p, " depth -= expanded.r * 256.0 * 256.0;\n"); WRITE(p, " expanded.g = floor(depth / 256.0);\n"); @@ -549,7 +549,7 @@ static void WriteZ16LEncoder(char*& p,API_TYPE ApiType) WRITE(p, " ocol0.r = expanded.b / 255.0;\n"); WRITE(p, " ocol0.a = expanded.g / 255.0;\n"); - WriteEncoderEnd(p, ApiType); + WriteEncoderEnd(p); } static void WriteZ24Encoder(char*& p, API_TYPE ApiType) @@ -568,7 +568,7 @@ static void WriteZ24Encoder(char*& p, API_TYPE ApiType) for (int i = 0; i < 2; i++) { - WRITE(p, " depth%i = clamp(depth%i * 16777216.0, 0.0, float(0xFFFFFF));\n", i, i); + WRITE(p, " depth%i *= 16777216.0;\n", i); WRITE(p, " expanded%i.r = floor(depth%i / (256.0 * 256.0));\n", i, i); WRITE(p, " depth%i -= expanded%i.r * 256.0 * 256.0;\n", i, i); @@ -591,7 +591,7 @@ static void WriteZ24Encoder(char*& p, API_TYPE ApiType) WRITE(p, " ocol0.a = expanded1.r / 255.0;\n"); WRITE(p, " }\n"); - WriteEncoderEnd(p, ApiType); + WriteEncoderEnd(p); } const char *GenerateEncodingShader(u32 format,API_TYPE ApiType) @@ -650,9 +650,11 @@ const char *GenerateEncodingShader(u32 format,API_TYPE ApiType) case GX_CTF_GB8: WriteCC8Encoder(p, "gb", ApiType); break; + case GX_CTF_Z8H: case GX_TF_Z8: WriteC8Encoder(p, "r", ApiType); break; + case GX_CTF_Z16R: case GX_TF_Z16: WriteZ16Encoder(p, ApiType); break; diff --git a/Source/Core/VideoCommon/TextureDecoder.h b/Source/Core/VideoCommon/TextureDecoder.h index 1abcc5705c..249024fa3a 100644 --- a/Source/Core/VideoCommon/TextureDecoder.h +++ b/Source/Core/VideoCommon/TextureDecoder.h @@ -12,10 +12,11 @@ enum TMEM_SIZE = 1024 * 1024, TMEM_LINE_SIZE = 32, }; -extern GC_ALIGNED16(u8 texMem[TMEM_SIZE]); +alignas(16) extern u8 texMem[TMEM_SIZE]; enum TextureFormat { + // These are the texture formats that can be read by the texture mapper. GX_TF_I4 = 0x0, GX_TF_I8 = 0x1, GX_TF_IA4 = 0x2, @@ -28,14 +29,21 @@ enum TextureFormat GX_TF_C14X2 = 0xA, GX_TF_CMPR = 0xE, - _GX_TF_CTF = 0x20, // copy-texture-format only (simply means linear?) - _GX_TF_ZTF = 0x10, // Z-texture-format + _GX_TF_ZTF = 0x10, // flag for Z texture formats (used internally by dolphin) - // these formats are also valid when copying targets + // Depth texture formats (which directly map to the equivalent colour format above.) + GX_TF_Z8 = 0x1 | _GX_TF_ZTF, + GX_TF_Z16 = 0x3 | _GX_TF_ZTF, + GX_TF_Z24X8 = 0x6 | _GX_TF_ZTF, + + _GX_TF_CTF = 0x20, // flag for copy-texture-format only (used internally by dolphin) + + // These are extra formats that can be used when copying from efb, + // they use one of texel formats from above, but pack diffrent data into them. GX_CTF_R4 = 0x0 | _GX_TF_CTF, GX_CTF_RA4 = 0x2 | _GX_TF_CTF, GX_CTF_RA8 = 0x3 | _GX_TF_CTF, - GX_CTF_YUVA8 = 0x6 | _GX_TF_CTF, + GX_CTF_YUVA8 = 0x6 | _GX_TF_CTF, // YUV 4:4:4 - Dolphin doesn't implement this format as no commercial games use it GX_CTF_A8 = 0x7 | _GX_TF_CTF, GX_CTF_R8 = 0x8 | _GX_TF_CTF, GX_CTF_G8 = 0x9 | _GX_TF_CTF, @@ -43,13 +51,12 @@ enum TextureFormat GX_CTF_RG8 = 0xB | _GX_TF_CTF, GX_CTF_GB8 = 0xC | _GX_TF_CTF, - GX_TF_Z8 = 0x1 | _GX_TF_ZTF, - GX_TF_Z16 = 0x3 | _GX_TF_ZTF, - GX_TF_Z24X8 = 0x6 | _GX_TF_ZTF, - + // extra depth texture formats that can be used for efb copies. GX_CTF_Z4 = 0x0 | _GX_TF_ZTF | _GX_TF_CTF, + GX_CTF_Z8H = 0x8 | _GX_TF_ZTF | _GX_TF_CTF, // This produces an identical result to to GX_TF_Z8 GX_CTF_Z8M = 0x9 | _GX_TF_ZTF | _GX_TF_CTF, GX_CTF_Z8L = 0xA | _GX_TF_ZTF | _GX_TF_CTF, + GX_CTF_Z16R = 0xB | _GX_TF_ZTF | _GX_TF_CTF, // Reversed version of GX_TF_Z16 GX_CTF_Z16L = 0xC | _GX_TF_ZTF | _GX_TF_CTF, }; diff --git a/Source/Core/VideoCommon/TextureDecoder_Common.cpp b/Source/Core/VideoCommon/TextureDecoder_Common.cpp index 092b8ec1e0..5f8f314a0b 100644 --- a/Source/Core/VideoCommon/TextureDecoder_Common.cpp +++ b/Source/Core/VideoCommon/TextureDecoder_Common.cpp @@ -5,8 +5,10 @@ #include <algorithm> #include <cmath> -#include "Common/Common.h" - +#include "Common/CommonFuncs.h" +#include "Common/CommonTypes.h" +#include "Common/MsgHandler.h" +#include "Common/Logging/Log.h" #include "VideoCommon/LookUpTables.h" #include "VideoCommon/sfont.inc" #include "VideoCommon/TextureDecoder.h" @@ -16,7 +18,7 @@ static bool TexFmt_Overlay_Center = false; // TRAM // STATE_TO_SAVE -GC_ALIGNED16(u8 texMem[TMEM_SIZE]); +alignas(16) u8 texMem[TMEM_SIZE]; int TexDecoder_GetTexelSizeInNibbles(int format) { @@ -35,7 +37,6 @@ int TexDecoder_GetTexelSizeInNibbles(int format) case GX_CTF_R4: return 1; case GX_CTF_RA4: return 2; case GX_CTF_RA8: return 4; - case GX_CTF_YUVA8: return 8; case GX_CTF_A8: return 2; case GX_CTF_R8: return 2; case GX_CTF_G8: return 2; @@ -48,10 +49,14 @@ int TexDecoder_GetTexelSizeInNibbles(int format) case GX_TF_Z24X8: return 8; case GX_CTF_Z4: return 1; + case GX_CTF_Z8H: return 2; case GX_CTF_Z8M: return 2; case GX_CTF_Z8L: return 2; + case GX_CTF_Z16R: return 4; case GX_CTF_Z16L: return 4; - default: return 1; + default: + PanicAlert("Unsupported Texture Format (%08x)! (GetTexelSizeInNibbles)", format); + return 1; } } @@ -88,11 +93,13 @@ int TexDecoder_GetBlockWidthInTexels(u32 format) case GX_TF_Z16: return 4; case GX_TF_Z24X8: return 4; case GX_CTF_Z4: return 8; + case GX_CTF_Z8H: return 8; case GX_CTF_Z8M: return 8; case GX_CTF_Z8L: return 8; + case GX_CTF_Z16R: return 4; case GX_CTF_Z16L: return 4; default: - ERROR_LOG(VIDEO, "Unsupported Texture Format (%08x)! (GetBlockWidthInTexels)", format); + PanicAlert("Unsupported Texture Format (%08x)! (GetBlockWidthInTexels)", format); return 8; } } @@ -125,11 +132,13 @@ int TexDecoder_GetBlockHeightInTexels(u32 format) case GX_TF_Z16: return 4; case GX_TF_Z24X8: return 4; case GX_CTF_Z4: return 8; + case GX_CTF_Z8H: return 4; case GX_CTF_Z8M: return 4; case GX_CTF_Z8L: return 4; + case GX_CTF_Z16R: return 4; case GX_CTF_Z16L: return 4; default: - ERROR_LOG(VIDEO, "Unsupported Texture Format (%08x)! (GetBlockHeightInTexels)", format); + PanicAlert("Unsupported Texture Format (%08x)! (GetBlockHeightInTexels)", format); return 4; } } diff --git a/Source/Core/VideoCommon/TextureDecoder_x64.cpp b/Source/Core/VideoCommon/TextureDecoder_x64.cpp index f68d35f842..3064b0f3c1 100644 --- a/Source/Core/VideoCommon/TextureDecoder_x64.cpp +++ b/Source/Core/VideoCommon/TextureDecoder_x64.cpp @@ -1065,8 +1065,8 @@ void _TexDecoder_DecodeImpl(u32 * dst, const u8 * src, int width, int height, in const __m128i dxt = _mm_loadu_si128((__m128i *)(src + sizeof(struct DXTBlock) * 2 * xStep)); // Copy the 2-bit indices from each DXT block: - GC_ALIGNED16( u32 dxttmp[4] ); - _mm_store_si128((__m128i *)dxttmp, dxt); + alignas(16) u32 dxttmp[4]; + _mm_store_si128((__m128i*)dxttmp, dxt); u32 dxt0sel = dxttmp[1]; u32 dxt1sel = dxttmp[3]; @@ -1204,10 +1204,10 @@ void _TexDecoder_DecodeImpl(u32 * dst, const u8 * src, int width, int height, in u32 *dst32 = ( dst + (y + z*4) * width + x ); // Copy the colors here: - GC_ALIGNED16( u32 colors0[4] ); - GC_ALIGNED16( u32 colors1[4] ); - _mm_store_si128((__m128i *)colors0, mmcolors0); - _mm_store_si128((__m128i *)colors1, mmcolors1); + alignas(16) u32 colors0[4]; + alignas(16) u32 colors1[4]; + _mm_store_si128((__m128i*)colors0, mmcolors0); + _mm_store_si128((__m128i*)colors1, mmcolors1); // Row 0: dst32[(width * 0) + 0] = colors0[(dxt0sel >> ((0*8)+6)) & 3]; diff --git a/Source/Core/VideoCommon/VertexLoader.cpp b/Source/Core/VideoCommon/VertexLoader.cpp index faf4599fdd..5673d1aa90 100644 --- a/Source/Core/VideoCommon/VertexLoader.cpp +++ b/Source/Core/VideoCommon/VertexLoader.cpp @@ -2,6 +2,7 @@ // Licensed under GPLv2+ // Refer to the license.txt file included. +#include "Common/Common.h" #include "Common/CommonTypes.h" #include "Common/MemoryUtil.h" diff --git a/Source/Core/VideoCommon/VertexLoaderARM64.cpp b/Source/Core/VideoCommon/VertexLoaderARM64.cpp index cf321baa60..4251acaaa0 100644 --- a/Source/Core/VideoCommon/VertexLoaderARM64.cpp +++ b/Source/Core/VideoCommon/VertexLoaderARM64.cpp @@ -21,7 +21,7 @@ ARM64Reg stride_reg = X11; ARM64Reg arraybase_reg = X10; ARM64Reg scale_reg = X9; -static const float GC_ALIGNED16(scale_factors[]) = +alignas(16) static const float scale_factors[] = { 1.0 / (1ULL << 0), 1.0 / (1ULL << 1), 1.0 / (1ULL << 2), 1.0 / (1ULL << 3), 1.0 / (1ULL << 4), 1.0 / (1ULL << 5), 1.0 / (1ULL << 6), 1.0 / (1ULL << 7), @@ -47,17 +47,36 @@ VertexLoaderARM64::VertexLoaderARM64(const TVtxDesc& vtx_desc, const VAT& vtx_at void VertexLoaderARM64::GetVertexAddr(int array, u64 attribute, ARM64Reg reg) { - ADD(reg, src_reg, m_src_ofs); if (attribute & MASK_INDEXED) { if (attribute == INDEX8) { - LDRB(INDEX_UNSIGNED, scratch1_reg, reg, 0); + if (m_src_ofs < 4096) + { + LDRB(INDEX_UNSIGNED, scratch1_reg, src_reg, m_src_ofs); + } + else + { + ADD(reg, src_reg, m_src_ofs); + LDRB(INDEX_UNSIGNED, scratch1_reg, reg, 0); + } m_src_ofs += 1; } else { - LDRH(INDEX_UNSIGNED, scratch1_reg, reg, 0); + if (m_src_ofs < 256) + { + LDURH(scratch1_reg, src_reg, m_src_ofs); + } + else if (m_src_ofs <= 8190 && !(m_src_ofs & 1)) + { + LDRH(INDEX_UNSIGNED, scratch1_reg, src_reg, m_src_ofs); + } + else + { + ADD(reg, src_reg, m_src_ofs); + LDRH(INDEX_UNSIGNED, scratch1_reg, reg, 0); + } m_src_ofs += 2; REV16(scratch1_reg, scratch1_reg); } @@ -74,6 +93,8 @@ void VertexLoaderARM64::GetVertexAddr(int array, u64 attribute, ARM64Reg reg) LDR(INDEX_UNSIGNED, EncodeRegTo64(scratch2_reg), arraybase_reg, array * 8); ADD(EncodeRegTo64(reg), EncodeRegTo64(scratch1_reg), EncodeRegTo64(scratch2_reg)); } + else + ADD(reg, src_reg, m_src_ofs); } s32 VertexLoaderARM64::GetAddressImm(int array, u64 attribute, Arm64Gen::ARM64Reg reg, u32 align) @@ -171,8 +192,7 @@ int VertexLoaderARM64::ReadVertex(u64 attribute, int format, int count_in, int c CMP(count_reg, 3); FixupBranch dont_store = B(CC_GT); MOVI2R(EncodeRegTo64(scratch2_reg), (u64)VertexLoaderManager::position_cache); - ORR(scratch1_reg, WSP, count_reg, ArithOption(count_reg, ST_LSL, 4)); - ADD(EncodeRegTo64(scratch1_reg), EncodeRegTo64(scratch1_reg), EncodeRegTo64(scratch2_reg)); + ADD(EncodeRegTo64(scratch1_reg), EncodeRegTo64(scratch2_reg), EncodeRegTo64(count_reg), ArithOption(EncodeRegTo64(count_reg), ST_LSL, 4)); m_float_emit.STUR(write_size, coords, EncodeRegTo64(scratch1_reg), -16); SetJumpTarget(dont_store); } @@ -347,12 +367,33 @@ void VertexLoaderARM64::GenerateVertexLoader() // We can touch all except v8-v15 // If we need to use those, we need to retain the lower 64bits(!) of the register - MOV(skipped_reg, WSP); + const u64 tc[8] = { + m_VtxDesc.Tex0Coord, m_VtxDesc.Tex1Coord, m_VtxDesc.Tex2Coord, m_VtxDesc.Tex3Coord, + m_VtxDesc.Tex4Coord, m_VtxDesc.Tex5Coord, m_VtxDesc.Tex6Coord, m_VtxDesc.Tex7Coord, + }; + + bool has_tc = false; + bool has_tc_scale = false; + for (int i = 0; i < 8; i++) + { + has_tc |= tc[i]; + has_tc_scale |= !!m_VtxAttr.texCoord[i].Frac; + } + + bool need_scale = (m_VtxAttr.ByteDequant && m_VtxAttr.PosFrac) || + (has_tc && has_tc_scale) || + m_VtxDesc.Normal; + + AlignCode16(); + if (m_VtxDesc.Position & MASK_INDEXED) + MOV(skipped_reg, WZR); MOV(saved_count, count_reg); MOVI2R(stride_reg, (u64)&g_main_cp_state.array_strides); MOVI2R(arraybase_reg, (u64)&VertexLoaderManager::cached_arraybases); - MOVI2R(scale_reg, (u64)&scale_factors); + + if (need_scale) + MOVI2R(scale_reg, (u64)&scale_factors); const u8* loop_start = GetCodePtr(); @@ -465,10 +506,7 @@ void VertexLoaderARM64::GenerateVertexLoader() } } - const u64 tc[8] = { - m_VtxDesc.Tex0Coord, m_VtxDesc.Tex1Coord, m_VtxDesc.Tex2Coord, m_VtxDesc.Tex3Coord, - m_VtxDesc.Tex4Coord, m_VtxDesc.Tex5Coord, m_VtxDesc.Tex6Coord, m_VtxDesc.Tex7Coord, - }; + for (int i = 0; i < 8; i++) { diff --git a/Source/Core/VideoCommon/VertexLoaderBase.cpp b/Source/Core/VideoCommon/VertexLoaderBase.cpp index d5831bc8e5..681f43b4fa 100644 --- a/Source/Core/VideoCommon/VertexLoaderBase.cpp +++ b/Source/Core/VideoCommon/VertexLoaderBase.cpp @@ -120,7 +120,7 @@ void VertexLoaderBase::AppendToString(std::string *dest) const i, m_VtxAttr.texCoord[i].Elements, posMode[tex_mode[i]], posFormats[m_VtxAttr.texCoord[i].Format])); } } - dest->append(StringFromFormat(" - %i v\n", m_numLoadedVertices)); + dest->append(StringFromFormat(" - %i v", m_numLoadedVertices)); } // a hacky implementation to compare two vertex loaders @@ -131,13 +131,14 @@ public: : VertexLoaderBase(vtx_desc, vtx_attr), a(_a), b(_b) { m_initialized = a && b && a->IsInitialized() && b->IsInitialized(); - bool can_test = a->m_VertexSize == b->m_VertexSize && - a->m_native_components == b->m_native_components && - a->m_native_vtx_decl.stride == b->m_native_vtx_decl.stride; if (m_initialized) { - if (can_test) + m_initialized = a->m_VertexSize == b->m_VertexSize && + a->m_native_components == b->m_native_components && + a->m_native_vtx_decl.stride == b->m_native_vtx_decl.stride; + + if (m_initialized) { m_VertexSize = a->m_VertexSize; m_native_components = a->m_native_components; @@ -152,8 +153,6 @@ public: b->m_VertexSize, b->m_native_components, b->m_native_vtx_decl.stride); } } - - m_initialized &= can_test; } ~VertexLoaderTester() override { diff --git a/Source/Core/VideoCommon/VertexLoaderManager.cpp b/Source/Core/VideoCommon/VertexLoaderManager.cpp index 46a0eeaab0..0099064ca5 100644 --- a/Source/Core/VideoCommon/VertexLoaderManager.cpp +++ b/Source/Core/VideoCommon/VertexLoaderManager.cpp @@ -109,7 +109,8 @@ void AppendListToString(std::string *dest) dest->reserve(dest->size() + total_size); for (const entry& entry : entries) { - dest->append(entry.text); + *dest += entry.text; + *dest += '\n'; } } diff --git a/Source/Core/VideoCommon/VertexLoaderUtils.h b/Source/Core/VideoCommon/VertexLoaderUtils.h index 143740ddf4..db37e26c43 100644 --- a/Source/Core/VideoCommon/VertexLoaderUtils.h +++ b/Source/Core/VideoCommon/VertexLoaderUtils.h @@ -4,6 +4,7 @@ #pragma once +#include <cstring> #include "Common/Common.h" #include "VideoCommon/VertexManagerBase.h" @@ -24,10 +25,11 @@ __forceinline void DataSkip() } template <typename T> -__forceinline T DataPeek(int _uOffset, u8** bufp = &g_video_buffer_read_ptr) +__forceinline T DataPeek(int _uOffset, u8* bufp = g_video_buffer_read_ptr) { - auto const result = Common::FromBigEndian(*reinterpret_cast<T*>(*bufp + _uOffset)); - return result; + T result; + std::memcpy(&result, &bufp[_uOffset], sizeof(T)); + return Common::FromBigEndian(result); } // TODO: kill these @@ -49,7 +51,7 @@ __forceinline u32 DataPeek32(int _uOffset) template <typename T> __forceinline T DataRead(u8** bufp = &g_video_buffer_read_ptr) { - auto const result = DataPeek<T>(0, bufp); + auto const result = DataPeek<T>(0, *bufp); *bufp += sizeof(T); return result; } @@ -77,9 +79,10 @@ __forceinline u32 DataReadU32() __forceinline u32 DataReadU32Unswapped() { - u32 tmp = *(u32*)g_video_buffer_read_ptr; - g_video_buffer_read_ptr += 4; - return tmp; + u32 result; + std::memcpy(&result, g_video_buffer_read_ptr, sizeof(u32)); + g_video_buffer_read_ptr += sizeof(u32); + return result; } __forceinline u8* DataGetPosition() @@ -90,6 +93,6 @@ __forceinline u8* DataGetPosition() template <typename T> __forceinline void DataWrite(T data) { - *(T*)g_vertex_manager_write_ptr = data; + std::memcpy(g_vertex_manager_write_ptr, &data, sizeof(T)); g_vertex_manager_write_ptr += sizeof(T); } diff --git a/Source/Core/VideoCommon/VertexLoaderX64.cpp b/Source/Core/VideoCommon/VertexLoaderX64.cpp index a298d7e1dd..d55ec17fef 100644 --- a/Source/Core/VideoCommon/VertexLoaderX64.cpp +++ b/Source/Core/VideoCommon/VertexLoaderX64.cpp @@ -38,7 +38,7 @@ VertexLoaderX64::VertexLoaderX64(const TVtxDesc& vtx_desc, const VAT& vtx_att) : if (!IsInitialized()) return; - AllocCodeSpace(4096); + AllocCodeSpace(4096, false); ClearCodeSpace(); GenerateVertexLoader(); WriteProtect(); @@ -53,21 +53,12 @@ OpArg VertexLoaderX64::GetVertexAddr(int array, u64 attribute) OpArg data = MDisp(src_reg, m_src_ofs); if (attribute & MASK_INDEXED) { - if (attribute == INDEX8) - { - MOVZX(64, 8, scratch1, data); - m_src_ofs += 1; - } - else - { - MOV(16, R(scratch1), data); - m_src_ofs += 2; - BSWAP(16, scratch1); - MOVZX(64, 16, scratch1, R(scratch1)); - } + int bits = attribute == INDEX8 ? 8 : 16; + LoadAndSwap(bits, scratch1, data); + m_src_ofs += bits / 8; if (array == ARRAY_POSITION) { - CMP(attribute == INDEX8 ? 8 : 16, R(scratch1), Imm8(-1)); + CMP(bits, R(scratch1), Imm8(-1)); m_skip_vertex = J_CC(CC_E, true); } IMUL(32, scratch1, MPIC(&g_main_cp_state.array_strides[array])); diff --git a/Source/Core/VideoCommon/VertexLoader_Color.cpp b/Source/Core/VideoCommon/VertexLoader_Color.cpp index df9d814357..0fbc639a53 100644 --- a/Source/Core/VideoCommon/VertexLoader_Color.cpp +++ b/Source/Core/VideoCommon/VertexLoader_Color.cpp @@ -2,6 +2,10 @@ // Licensed under GPLv2+ // Refer to the license.txt file included. +#include <cstring> + +#include "Common/Common.h" +#include "Common/CommonFuncs.h" #include "Common/CommonTypes.h" #include "VideoCommon/VertexLoader.h" @@ -12,15 +16,15 @@ #define AMASK 0xFF000000 -__forceinline void _SetCol(VertexLoader* loader, u32 val) +static void SetCol(VertexLoader* loader, u32 val) { DataWrite(val); loader->m_colIndex++; } -//color comes in format BARG in 16 bits -//BARG -> AABBGGRR -__forceinline void _SetCol4444(VertexLoader* loader, u16 val_) +// Color comes in format BARG in 16 bits +// BARG -> AABBGGRR +static void SetCol4444(VertexLoader* loader, u16 val_) { u32 col, val = val_; col = val & 0x00F0; // col = 000000R0; @@ -28,24 +32,24 @@ __forceinline void _SetCol4444(VertexLoader* loader, u16 val_) col |= (val & 0xF000) << 8; // col |= 00B00000; col |= (val & 0x0F00) << 20; // col |= A0000000; col |= col >> 4; // col = A0B0G0R0 | 0A0B0G0R; - _SetCol(loader, col); + SetCol(loader, col); } -//color comes in format RGBA -//RRRRRRGG GGGGBBBB BBAAAAAA -__forceinline void _SetCol6666(VertexLoader* loader, u32 val) +// Color comes in format RGBA +// RRRRRRGG GGGGBBBB BBAAAAAA +static void SetCol6666(VertexLoader* loader, u32 val) { u32 col = (val >> 16) & 0x000000FC; col |= (val >> 2) & 0x0000FC00; col |= (val << 12) & 0x00FC0000; col |= (val << 26) & 0xFC000000; col |= (col >> 6) & 0x03030303; - _SetCol(loader, col); + SetCol(loader, col); } -//color comes in RGB -//RRRRRGGG GGGBBBBB -__forceinline void _SetCol565(VertexLoader* loader, u16 val_) +// Color comes in RGB +// RRRRRGGG GGGBBBBB +static void SetCol565(VertexLoader* loader, u16 val_) { u32 col, val = val_; col = (val >> 8) & 0x0000F8; @@ -53,56 +57,64 @@ __forceinline void _SetCol565(VertexLoader* loader, u16 val_) col |= (val << 19) & 0xF80000; col |= (col >> 5) & 0x070007; col |= (col >> 6) & 0x000300; - _SetCol(loader, col | AMASK); + SetCol(loader, col | AMASK); } -__forceinline u32 _Read24(const u8 *addr) +static u32 Read32(const u8* addr) { - return (*(const u32 *)addr) | AMASK; + u32 value; + std::memcpy(&value, addr, sizeof(u32)); + return value; } -__forceinline u32 _Read32(const u8 *addr) +static u32 Read24(const u8* addr) { - return *(const u32 *)addr; + return Read32(addr) | AMASK; } - void Color_ReadDirect_24b_888(VertexLoader* loader) { - _SetCol(loader, _Read24(DataGetPosition())); + SetCol(loader, Read24(DataGetPosition())); DataSkip(3); } void Color_ReadDirect_32b_888x(VertexLoader* loader) { - _SetCol(loader, _Read24(DataGetPosition())); + SetCol(loader, Read24(DataGetPosition())); DataSkip(4); } void Color_ReadDirect_16b_565(VertexLoader* loader) { - _SetCol565(loader, DataReadU16()); + SetCol565(loader, DataReadU16()); } void Color_ReadDirect_16b_4444(VertexLoader* loader) { - _SetCol4444(loader, *(u16*)DataGetPosition()); + u16 value; + std::memcpy(&value, DataGetPosition(), sizeof(u16)); + + SetCol4444(loader, value); DataSkip(2); } void Color_ReadDirect_24b_6666(VertexLoader* loader) { - _SetCol6666(loader, Common::swap32(DataGetPosition() - 1)); + SetCol6666(loader, Common::swap32(DataGetPosition() - 1)); DataSkip(3); } void Color_ReadDirect_32b_8888(VertexLoader* loader) { - _SetCol(loader, DataReadU32Unswapped()); + SetCol(loader, DataReadU32Unswapped()); } template <typename I> void Color_ReadIndex_16b_565(VertexLoader* loader) { auto const Index = DataRead<I>(); - u16 val = Common::swap16(*(const u16 *)(VertexLoaderManager::cached_arraybases[ARRAY_COLOR + loader->m_colIndex] + (Index * g_main_cp_state.array_strides[ARRAY_COLOR + loader->m_colIndex]))); - _SetCol565(loader, val); + const u8* const address = VertexLoaderManager::cached_arraybases[ARRAY_COLOR + loader->m_colIndex] + (Index * g_main_cp_state.array_strides[ARRAY_COLOR + loader->m_colIndex]); + + u16 value; + std::memcpy(&value, address, sizeof(u16)); + + SetCol565(loader, Common::swap16(value)); } template <typename I> @@ -110,7 +122,7 @@ void Color_ReadIndex_24b_888(VertexLoader* loader) { auto const Index = DataRead<I>(); const u8 *iAddress = VertexLoaderManager::cached_arraybases[ARRAY_COLOR + loader->m_colIndex] + (Index * g_main_cp_state.array_strides[ARRAY_COLOR + loader->m_colIndex]); - _SetCol(loader, _Read24(iAddress)); + SetCol(loader, Read24(iAddress)); } template <typename I> @@ -118,15 +130,19 @@ void Color_ReadIndex_32b_888x(VertexLoader* loader) { auto const Index = DataRead<I>(); const u8 *iAddress = VertexLoaderManager::cached_arraybases[ARRAY_COLOR + loader->m_colIndex] + (Index * g_main_cp_state.array_strides[ARRAY_COLOR + loader->m_colIndex]); - _SetCol(loader, _Read24(iAddress)); + SetCol(loader, Read24(iAddress)); } template <typename I> void Color_ReadIndex_16b_4444(VertexLoader* loader) { auto const Index = DataRead<I>(); - u16 val = *(const u16 *)(VertexLoaderManager::cached_arraybases[ARRAY_COLOR + loader->m_colIndex] + (Index * g_main_cp_state.array_strides[ARRAY_COLOR + loader->m_colIndex])); - _SetCol4444(loader, val); + const u8* const address = VertexLoaderManager::cached_arraybases[ARRAY_COLOR + loader->m_colIndex] + (Index * g_main_cp_state.array_strides[ARRAY_COLOR + loader->m_colIndex]); + + u16 value; + std::memcpy(&value, address, sizeof(u16)); + + SetCol4444(loader, value); } template <typename I> @@ -135,7 +151,7 @@ void Color_ReadIndex_24b_6666(VertexLoader* loader) auto const Index = DataRead<I>(); const u8* pData = VertexLoaderManager::cached_arraybases[ARRAY_COLOR + loader->m_colIndex] + (Index * g_main_cp_state.array_strides[ARRAY_COLOR + loader->m_colIndex]) - 1; u32 val = Common::swap32(pData); - _SetCol6666(loader, val); + SetCol6666(loader, val); } template <typename I> @@ -143,7 +159,7 @@ void Color_ReadIndex_32b_8888(VertexLoader* loader) { auto const Index = DataRead<I>(); const u8 *iAddress = VertexLoaderManager::cached_arraybases[ARRAY_COLOR + loader->m_colIndex] + (Index * g_main_cp_state.array_strides[ARRAY_COLOR + loader->m_colIndex]); - _SetCol(loader, _Read32(iAddress)); + SetCol(loader, Read32(iAddress)); } void Color_ReadIndex8_16b_565(VertexLoader* loader) { Color_ReadIndex_16b_565<u8>(loader); } diff --git a/Source/Core/VideoCommon/VertexLoader_Normal.cpp b/Source/Core/VideoCommon/VertexLoader_Normal.cpp index 615a09c3a9..5561d51c7a 100644 --- a/Source/Core/VideoCommon/VertexLoader_Normal.cpp +++ b/Source/Core/VideoCommon/VertexLoader_Normal.cpp @@ -5,6 +5,7 @@ #include <cmath> #include <type_traits> +#include "Common/Common.h" #include "Common/CommonTypes.h" #include "VideoCommon/VertexLoader.h" #include "VideoCommon/VertexLoader_Normal.h" diff --git a/Source/Core/VideoCommon/VertexLoader_Position.cpp b/Source/Core/VideoCommon/VertexLoader_Position.cpp index 906cf61d71..609038ab3e 100644 --- a/Source/Core/VideoCommon/VertexLoader_Position.cpp +++ b/Source/Core/VideoCommon/VertexLoader_Position.cpp @@ -4,6 +4,7 @@ #include <type_traits> +#include "Common/Common.h" #include "Common/CommonTypes.h" #include "VideoCommon/VertexLoader.h" #include "VideoCommon/VertexLoader_Position.h" diff --git a/Source/Core/VideoCommon/VertexLoader_Position.h b/Source/Core/VideoCommon/VertexLoader_Position.h index 65a97e5bde..306a1a8c4a 100644 --- a/Source/Core/VideoCommon/VertexLoader_Position.h +++ b/Source/Core/VideoCommon/VertexLoader_Position.h @@ -6,12 +6,9 @@ #include "VideoCommon/NativeVertexFormat.h" -class VertexLoader_Position { +class VertexLoader_Position +{ public: - - // Init - static void Init(); - // GetSize static unsigned int GetSize(u64 _type, unsigned int _format, unsigned int _elements); diff --git a/Source/Core/VideoCommon/VertexLoader_TextCoord.cpp b/Source/Core/VideoCommon/VertexLoader_TextCoord.cpp index 28a9a33871..764efe7279 100644 --- a/Source/Core/VideoCommon/VertexLoader_TextCoord.cpp +++ b/Source/Core/VideoCommon/VertexLoader_TextCoord.cpp @@ -4,6 +4,7 @@ #include <type_traits> +#include "Common/Common.h" #include "Common/CommonTypes.h" #include "VideoCommon/VertexLoader.h" #include "VideoCommon/VertexLoader_TextCoord.h" diff --git a/Source/Core/VideoCommon/VertexLoader_TextCoord.h b/Source/Core/VideoCommon/VertexLoader_TextCoord.h index 0e3c7bac62..4d8d352058 100644 --- a/Source/Core/VideoCommon/VertexLoader_TextCoord.h +++ b/Source/Core/VideoCommon/VertexLoader_TextCoord.h @@ -9,10 +9,6 @@ class VertexLoader_TextCoord { public: - - // Init - static void Init(); - // GetSize static unsigned int GetSize(u64 _type, unsigned int _format, unsigned int _elements); diff --git a/Source/Core/VideoCommon/VertexManagerBase.cpp b/Source/Core/VideoCommon/VertexManagerBase.cpp index e9e8ebf901..6358ed371c 100644 --- a/Source/Core/VideoCommon/VertexManagerBase.cpp +++ b/Source/Core/VideoCommon/VertexManagerBase.cpp @@ -250,8 +250,7 @@ void VertexManager::Flush() GeometryShaderManager::SetConstants(); PixelShaderManager::SetConstants(); - bool useDstAlpha = !g_ActiveConfig.bDstAlphaPass && - bpmem.dstalpha.enable && + bool useDstAlpha = bpmem.dstalpha.enable && bpmem.blendmode.alphaupdate && bpmem.zcontrol.pixel_format == PEControl::RGBA6_Z24; diff --git a/Source/Core/VideoCommon/VertexShaderGen.cpp b/Source/Core/VideoCommon/VertexShaderGen.cpp index 97ac15e887..63f80488e4 100644 --- a/Source/Core/VideoCommon/VertexShaderGen.cpp +++ b/Source/Core/VideoCommon/VertexShaderGen.cpp @@ -32,29 +32,6 @@ static inline void GenerateVertexShader(T& out, u32 components, API_TYPE api_typ _assert_(bpmem.genMode.numtexgens == xfmem.numTexGen.numTexGens); _assert_(bpmem.genMode.numcolchans == xfmem.numChan.numColorChans); - if (DriverDetails::HasBug(DriverDetails::BUG_BROKENIVECSHIFTS)) - { - // Add functions to do shifts on scalars and ivecs. - // This is included in the vertex shader for lighting shader generation. - out.Write("int ilshift(int a, int b) { return a << b; }\n" - "int irshift(int a, int b) { return a >> b; }\n" - - "int2 ilshift(int2 a, int2 b) { return int2(a.x << b.x, a.y << b.y); }\n" - "int2 ilshift(int2 a, int b) { return int2(a.x << b, a.y << b); }\n" - "int2 irshift(int2 a, int2 b) { return int2(a.x >> b.x, a.y >> b.y); }\n" - "int2 irshift(int2 a, int b) { return int2(a.x >> b, a.y >> b); }\n" - - "int3 ilshift(int3 a, int3 b) { return int3(a.x << b.x, a.y << b.y, a.z << b.z); }\n" - "int3 ilshift(int3 a, int b) { return int3(a.x << b, a.y << b, a.z << b); }\n" - "int3 irshift(int3 a, int3 b) { return int3(a.x >> b.x, a.y >> b.y, a.z >> b.z); }\n" - "int3 irshift(int3 a, int b) { return int3(a.x >> b, a.y >> b, a.z >> b); }\n" - - "int4 ilshift(int4 a, int4 b) { return int4(a.x << b.x, a.y << b.y, a.z << b.z, a.w << b.w); }\n" - "int4 ilshift(int4 a, int b) { return int4(a.x << b, a.y << b, a.z << b, a.w << b); }\n" - "int4 irshift(int4 a, int4 b) { return int4(a.x >> b.x, a.y >> b.y, a.z >> b.z, a.w >> b.w); }\n" - "int4 irshift(int4 a, int b) { return int4(a.x >> b, a.y >> b, a.z >> b, a.w >> b); }\n\n"); - } - out.Write("%s", s_lighting_struct); // uniforms @@ -100,7 +77,7 @@ static inline void GenerateVertexShader(T& out, u32 components, API_TYPE api_typ if (g_ActiveConfig.backend_info.bSupportsGeometryShaders) { out.Write("out VertexData {\n"); - GenerateVSOutputMembers<T>(out, api_type, g_ActiveConfig.backend_info.bSupportsBindingLayout ? "centroid" : "centroid out"); + GenerateVSOutputMembers<T>(out, api_type, GetInterpolationQualifier(api_type, false, true)); out.Write("} vs;\n"); } else @@ -110,17 +87,17 @@ static inline void GenerateVertexShader(T& out, u32 components, API_TYPE api_typ { if (i < xfmem.numTexGen.numTexGens) { - out.Write("centroid out float3 uv%d;\n", i); + out.Write("%s out float3 uv%d;\n", GetInterpolationQualifier(api_type), i); } } - out.Write("centroid out float4 clipPos;\n"); + out.Write("%s out float4 clipPos;\n", GetInterpolationQualifier(api_type)); if (g_ActiveConfig.bEnablePixelLighting) { - out.Write("centroid out float3 Normal;\n"); - out.Write("centroid out float3 WorldPos;\n"); + out.Write("%s out float3 Normal;\n", GetInterpolationQualifier(api_type)); + out.Write("%s out float3 WorldPos;\n", GetInterpolationQualifier(api_type)); } - out.Write("centroid out float4 colors_0;\n"); - out.Write("centroid out float4 colors_1;\n"); + out.Write("%s out float4 colors_0;\n", GetInterpolationQualifier(api_type)); + out.Write("%s out float4 colors_1;\n", GetInterpolationQualifier(api_type)); } out.Write("void main()\n{\n"); @@ -156,22 +133,12 @@ static inline void GenerateVertexShader(T& out, u32 components, API_TYPE api_typ // transforms if (components & VB_HAS_POSMTXIDX) { - if (is_writing_shadercode && (DriverDetails::HasBug(DriverDetails::BUG_NODYNUBOACCESS) && !DriverDetails::HasBug(DriverDetails::BUG_ANNIHILATEDUBOS))) - { - // This'll cause issues, but it can't be helped - out.Write("float4 pos = float4(dot(" I_TRANSFORMMATRICES"[0], rawpos), dot(" I_TRANSFORMMATRICES"[1], rawpos), dot(" I_TRANSFORMMATRICES"[2], rawpos), 1);\n"); - if (components & VB_HAS_NRMALL) - out.Write("float3 N0 = " I_NORMALMATRICES"[0].xyz, N1 = " I_NORMALMATRICES"[1].xyz, N2 = " I_NORMALMATRICES"[2].xyz;\n"); - } - else - { - out.Write("float4 pos = float4(dot(" I_TRANSFORMMATRICES"[posmtx], rawpos), dot(" I_TRANSFORMMATRICES"[posmtx+1], rawpos), dot(" I_TRANSFORMMATRICES"[posmtx+2], rawpos), 1);\n"); + out.Write("float4 pos = float4(dot(" I_TRANSFORMMATRICES"[posmtx], rawpos), dot(" I_TRANSFORMMATRICES"[posmtx+1], rawpos), dot(" I_TRANSFORMMATRICES"[posmtx+2], rawpos), 1);\n"); - if (components & VB_HAS_NRMALL) - { - out.Write("int normidx = posmtx >= 32 ? (posmtx-32) : posmtx;\n"); - out.Write("float3 N0 = " I_NORMALMATRICES"[normidx].xyz, N1 = " I_NORMALMATRICES"[normidx+1].xyz, N2 = " I_NORMALMATRICES"[normidx+2].xyz;\n"); - } + if (components & VB_HAS_NRMALL) + { + out.Write("int normidx = posmtx >= 32 ? (posmtx-32) : posmtx;\n"); + out.Write("float3 N0 = " I_NORMALMATRICES"[normidx].xyz, N1 = " I_NORMALMATRICES"[normidx+1].xyz, N2 = " I_NORMALMATRICES"[normidx+2].xyz;\n"); } if (components & VB_HAS_NRM0) @@ -241,7 +208,8 @@ static inline void GenerateVertexShader(T& out, u32 components, API_TYPE api_typ switch (texinfo.sourcerow) { case XF_SRCGEOM_INROW: - _assert_(texinfo.inputform == XF_TEXINPUT_ABC1); + // The following assert was triggered in Super Smash Bros. Project M 3.6. + //_assert_(texinfo.inputform == XF_TEXINPUT_ABC1); out.Write("coord = rawpos;\n"); // pos.w is 1 break; case XF_SRCNORMAL_INROW: @@ -291,7 +259,8 @@ static inline void GenerateVertexShader(T& out, u32 components, API_TYPE api_typ } else { - _assert_(0); // should have normals + // The following assert was triggered in House of the Dead Overkill and Star Wars Rogue Squadron 2 + //_assert_(0); // should have normals uid_data->texMtxInfo[i].embosssourceshift = xfmem.texMtxInfo[i].embosssourceshift; out.Write("o.tex%d.xyz = o.tex%d.xyz;\n", i, texinfo.embosssourceshift); } diff --git a/Source/Core/VideoCommon/VertexShaderManager.cpp b/Source/Core/VideoCommon/VertexShaderManager.cpp index 509974c373..93d248938b 100644 --- a/Source/Core/VideoCommon/VertexShaderManager.cpp +++ b/Source/Core/VideoCommon/VertexShaderManager.cpp @@ -9,6 +9,8 @@ #include "Common/BitSet.h" #include "Common/CommonTypes.h" #include "Common/MathUtil.h" +#include "Core/ConfigManager.h" +#include "Core/Core.h" #include "VideoCommon/BPMemory.h" #include "VideoCommon/CPMemory.h" #include "VideoCommon/RenderBase.h" @@ -20,7 +22,7 @@ #include "VideoCommon/VideoConfig.h" #include "VideoCommon/XFMemory.h" -static float GC_ALIGNED16(g_fProjectionMatrix[16]); +alignas(16) static float g_fProjectionMatrix[16]; // track changes static bool bTexMatricesChanged[2], bPosNormalMatrixChanged, bProjectionChanged, bViewportChanged; @@ -87,6 +89,23 @@ static float PHackValue(std::string sValue) return f; } +// Due to the BT.601 standard which the GameCube is based on being a compromise +// between PAL and NTSC, neither standard gets square pixels. They are each off +// by ~9% in opposite directions. +// Just in case any game decides to take this into account, we do both these +// tests with a large amount of slop. +static bool AspectIs4_3(float width, float height) +{ + float aspect = fabsf(width / height); + return fabsf(aspect - 4.0f / 3.0f) < 4.0f / 3.0f * 0.11; // within 11% of 4:3 +} + +static bool AspectIs16_9(float width, float height) +{ + float aspect = fabsf(width / height); + return fabsf(aspect - 16.0f / 9.0f) < 16.0f / 9.0f * 0.11; // within 11% of 16:9 +} + void UpdateProjectionHack(int iPhackvalue[], std::string sPhackvalue[]) { float fhackvalue1 = 0, fhackvalue2 = 0; @@ -417,6 +436,16 @@ void VertexShaderManager::SetConstants() g_fProjectionMatrix[14] = -1.0f; g_fProjectionMatrix[15] = 0.0f; + // Heuristic to detect if a GameCube game is in 16:9 anamorphic widescreen mode. + if (!SConfig::GetInstance().bWii) + { + bool viewport_is_4_3 = AspectIs4_3(xfmem.viewport.wd, xfmem.viewport.ht); + if (AspectIs16_9(rawProjection[2], rawProjection[0]) && viewport_is_4_3) + g_aspect_wide = true; // Projection is 16:9 and viewport is 4:3, we are rendering an anamorphic widescreen picture + else if (AspectIs4_3(rawProjection[2], rawProjection[0]) && viewport_is_4_3) + g_aspect_wide = false; // Project and viewports are both 4:3, we are rendering a normal image. + } + SETSTAT_FT(stats.gproj_0, g_fProjectionMatrix[0]); SETSTAT_FT(stats.gproj_1, g_fProjectionMatrix[1]); SETSTAT_FT(stats.gproj_2, g_fProjectionMatrix[2]); @@ -654,7 +683,7 @@ void VertexShaderManager::SetProjectionChanged() bProjectionChanged = true; } -void VertexShaderManager::SetMaterialColorChanged(int index, u32 color) +void VertexShaderManager::SetMaterialColorChanged(int index) { nMaterialsChanged[index] = true; } diff --git a/Source/Core/VideoCommon/VertexShaderManager.h b/Source/Core/VideoCommon/VertexShaderManager.h index 325b18a2cd..35c146c7aa 100644 --- a/Source/Core/VideoCommon/VertexShaderManager.h +++ b/Source/Core/VideoCommon/VertexShaderManager.h @@ -28,7 +28,7 @@ public: static void SetTexMatrixChangedB(u32 value); static void SetViewportChanged(); static void SetProjectionChanged(); - static void SetMaterialColorChanged(int index, u32 color); + static void SetMaterialColorChanged(int index); static void TranslateView(float x, float y, float z = 0.0f); static void RotateView(float x, float y); diff --git a/Source/Core/VideoCommon/VideoBackendBase.cpp b/Source/Core/VideoCommon/VideoBackendBase.cpp index c0e466cb5c..69beee3f1d 100644 --- a/Source/Core/VideoCommon/VideoBackendBase.cpp +++ b/Source/Core/VideoCommon/VideoBackendBase.cpp @@ -18,21 +18,6 @@ static VideoBackend* s_default_backend = nullptr; #ifdef _WIN32 #include <windows.h> -// http://msdn.microsoft.com/en-us/library/ms725491.aspx -static bool IsGteVista() -{ - OSVERSIONINFOEX osvi; - DWORDLONG dwlConditionMask = 0; - - ZeroMemory(&osvi, sizeof(OSVERSIONINFOEX)); - osvi.dwOSVersionInfoSize = sizeof(OSVERSIONINFOEX); - osvi.dwMajorVersion = 6; - - VER_SET_CONDITION(dwlConditionMask, VER_MAJORVERSION, VER_GREATER_EQUAL); - - return VerifyVersionInfo(&osvi, VER_MAJORVERSION, dwlConditionMask) != FALSE; -} - // Nvidia drivers >= v302 will check if the application exports a global // variable named NvOptimusEnablement to know if it should run the app in high // performance graphics mode or using the IGP. @@ -48,8 +33,7 @@ void VideoBackend::PopulateList() // OGL > D3D11 > SW g_available_video_backends.push_back(backends[0] = new OGL::VideoBackend); #ifdef _WIN32 - if (IsGteVista()) - g_available_video_backends.push_back(backends[1] = new DX11::VideoBackend); + g_available_video_backends.push_back(backends[1] = new DX11::VideoBackend); #endif g_available_video_backends.push_back(backends[3] = new SW::VideoSoftware); diff --git a/Source/Core/VideoCommon/VideoBackendBase.h b/Source/Core/VideoCommon/VideoBackendBase.h index 4671afdc6b..f6cd1d12bf 100644 --- a/Source/Core/VideoCommon/VideoBackendBase.h +++ b/Source/Core/VideoCommon/VideoBackendBase.h @@ -15,9 +15,8 @@ namespace MMIO { class Mapping; } enum FieldType { - FIELD_PROGRESSIVE = 0, - FIELD_UPPER, - FIELD_LOWER + FIELD_ODD = 0, + FIELD_EVEN = 1, }; enum EFBAccessType @@ -161,5 +160,4 @@ public: protected: void InitializeShared(); - void InvalidState(); }; diff --git a/Source/Core/VideoCommon/VideoCommon.h b/Source/Core/VideoCommon/VideoCommon.h index b81ebb8929..ba339e5c4f 100644 --- a/Source/Core/VideoCommon/VideoCommon.h +++ b/Source/Core/VideoCommon/VideoCommon.h @@ -12,6 +12,9 @@ #include "Common/MathUtil.h" #include "VideoCommon/VideoBackendBase.h" +// Global flag to signal if FifoRecorder is active. +extern bool g_bRecordFifoData; + // These are accurate (disregarding AA modes). enum { @@ -19,9 +22,10 @@ enum EFB_HEIGHT = 528, }; -// XFB width is decided by EFB copy operation. The VI can do horizontal -// scaling (TODO: emulate). -const u32 MAX_XFB_WIDTH = EFB_WIDTH; +// Max XFB width is 720. You can only copy out 640 wide areas of efb to XFB +// so you need multiple copies to do the full width. +// The VI can do horizontal scaling (TODO: emulate). +const u32 MAX_XFB_WIDTH = 720; // Although EFB height is 528, 574-line XFB's can be created either with // vertical scaling by the EFB copy operation or copying to multiple XFB's @@ -65,12 +69,12 @@ struct TargetRectangle : public MathUtil::Rectangle<int> #define LOG_VTX() -typedef enum +enum API_TYPE { API_OPENGL = 1, API_D3D = 2, API_NONE = 3 -} API_TYPE; +}; inline u32 RGBA8ToRGBA6ToRGBA8(u32 src) { diff --git a/Source/Core/VideoCommon/VideoCommon.vcxproj b/Source/Core/VideoCommon/VideoCommon.vcxproj index ff56a97703..98eecf8542 100644 --- a/Source/Core/VideoCommon/VideoCommon.vcxproj +++ b/Source/Core/VideoCommon/VideoCommon.vcxproj @@ -1,5 +1,5 @@ <?xml version="1.0" encoding="utf-8"?> -<Project DefaultTargets="Build" ToolsVersion="12.0" xmlns="http://schemas.microsoft.com/developer/msbuild/2003"> +<Project DefaultTargets="Build" ToolsVersion="14.0" xmlns="http://schemas.microsoft.com/developer/msbuild/2003"> <ItemGroup Label="ProjectConfigurations"> <ProjectConfiguration Include="Debug|x64"> <Configuration>Debug</Configuration> @@ -16,7 +16,7 @@ <Import Project="$(VCTargetsPath)\Microsoft.Cpp.Default.props" /> <PropertyGroup Label="Configuration"> <ConfigurationType>StaticLibrary</ConfigurationType> - <PlatformToolset>v120</PlatformToolset> + <PlatformToolset>v140</PlatformToolset> <CharacterSet>Unicode</CharacterSet> </PropertyGroup> <PropertyGroup Condition="'$(Configuration)'=='Debug'" Label="Configuration"> @@ -162,4 +162,4 @@ <Import Project="$(VCTargetsPath)\Microsoft.Cpp.targets" /> <ImportGroup Label="ExtensionTargets"> </ImportGroup> -</Project> +</Project>
\ No newline at end of file diff --git a/Source/Core/VideoCommon/VideoConfig.cpp b/Source/Core/VideoCommon/VideoConfig.cpp index 92abca4819..6aadfbf075 100644 --- a/Source/Core/VideoCommon/VideoConfig.cpp +++ b/Source/Core/VideoCommon/VideoConfig.cpp @@ -78,8 +78,8 @@ void VideoConfig::Load(const std::string& ini_file) settings->Get("EnablePixelLighting", &bEnablePixelLighting, 0); settings->Get("FastDepthCalc", &bFastDepthCalc, true); settings->Get("MSAA", &iMultisampleMode, 0); + settings->Get("SSAA", &bSSAA, false); settings->Get("EFBScale", &iEFBScale, (int)SCALE_1X); // native - settings->Get("DstAlphaPass", &bDstAlphaPass, false); settings->Get("TexFmtOverlayEnable", &bTexFmtOverlayEnable, 0); settings->Get("TexFmtOverlayCenter", &bTexFmtOverlayCenter, 0); settings->Get("WireFrame", &bWireFrame, 0); @@ -99,6 +99,7 @@ void VideoConfig::Load(const std::string& ini_file) IniFile::Section* hacks = iniFile.GetOrCreateSection("Hacks"); hacks->Get("EFBAccessEnable", &bEFBAccessEnable, true); hacks->Get("BBoxEnable", &bBBoxEnable, false); + hacks->Get("ForceProgressive", &bForceProgressive, true); hacks->Get("EFBToTextureEnable", &bSkipEFBCopyToRam, true); hacks->Get("EFBScaledCopy", &bCopyEFBScaled, true); hacks->Get("EFBEmulateFormatChanges", &bEFBEmulateFormatChanges, false); @@ -160,6 +161,8 @@ void VideoConfig::GameIniLoad() CHECK_SETTING("Video_Settings", "EnablePixelLighting", bEnablePixelLighting); CHECK_SETTING("Video_Settings", "FastDepthCalc", bFastDepthCalc); CHECK_SETTING("Video_Settings", "MSAA", iMultisampleMode); + CHECK_SETTING("Video_Settings", "SSAA", bSSAA); + int tmp = -9000; CHECK_SETTING("Video_Settings", "EFBScale", tmp); // integral if (tmp != -9000) @@ -187,7 +190,6 @@ void VideoConfig::GameIniLoad() } } - CHECK_SETTING("Video_Settings", "DstAlphaPass", bDstAlphaPass); CHECK_SETTING("Video_Settings", "DisableFog", bDisableFog); CHECK_SETTING("Video_Enhancements", "ForceFiltering", bForceFiltering); @@ -204,6 +206,7 @@ void VideoConfig::GameIniLoad() CHECK_SETTING("Video_Hacks", "EFBAccessEnable", bEFBAccessEnable); CHECK_SETTING("Video_Hacks", "BBoxEnable", bBBoxEnable); + CHECK_SETTING("Video_Hacks", "ForceProgressive", bForceProgressive); CHECK_SETTING("Video_Hacks", "EFBToTextureEnable", bSkipEFBCopyToRam); CHECK_SETTING("Video_Hacks", "EFBScaledCopy", bCopyEFBScaled); CHECK_SETTING("Video_Hacks", "EFBEmulateFormatChanges", bEFBEmulateFormatChanges); @@ -272,11 +275,11 @@ void VideoConfig::Save(const std::string& ini_file) settings->Set("FastDepthCalc", bFastDepthCalc); settings->Set("ShowEFBCopyRegions", bShowEFBCopyRegions); settings->Set("MSAA", iMultisampleMode); + settings->Set("SSAA", bSSAA); settings->Set("EFBScale", iEFBScale); settings->Set("TexFmtOverlayEnable", bTexFmtOverlayEnable); settings->Set("TexFmtOverlayCenter", bTexFmtOverlayCenter); settings->Set("Wireframe", bWireFrame); - settings->Set("DstAlphaPass", bDstAlphaPass); settings->Set("DisableFog", bDisableFog); settings->Set("EnableShaderDebugging", bEnableShaderDebugging); settings->Set("BorderlessFullscreen", bBorderlessFullscreen); @@ -293,6 +296,7 @@ void VideoConfig::Save(const std::string& ini_file) IniFile::Section* hacks = iniFile.GetOrCreateSection("Hacks"); hacks->Set("EFBAccessEnable", bEFBAccessEnable); hacks->Set("BBoxEnable", bBBoxEnable); + hacks->Set("ForceProgressive", bForceProgressive); hacks->Set("EFBToTextureEnable", bSkipEFBCopyToRam); hacks->Set("EFBScaledCopy", bCopyEFBScaled); hacks->Set("EFBEmulateFormatChanges", bEFBEmulateFormatChanges); diff --git a/Source/Core/VideoCommon/VideoConfig.h b/Source/Core/VideoCommon/VideoConfig.h index 13451bbf5a..f6b4f9e7f0 100644 --- a/Source/Core/VideoCommon/VideoConfig.h +++ b/Source/Core/VideoCommon/VideoConfig.h @@ -25,10 +25,10 @@ enum AspectMode { - ASPECT_AUTO = 0, - ASPECT_FORCE_16_9 = 1, - ASPECT_FORCE_4_3 = 2, - ASPECT_STRETCH = 3, + ASPECT_AUTO = 0, + ASPECT_ANALOG_WIDE = 1, + ASPECT_ANALOG = 2, + ASPECT_STRETCH = 3, }; enum EFBScale @@ -75,6 +75,7 @@ struct VideoConfig final // Enhancements int iMultisampleMode; + bool bSSAA; int iEFBScale; bool bForceFiltering; int iMaxAnisotropy; @@ -95,7 +96,6 @@ struct VideoConfig final // Render bool bWireFrame; - bool bDstAlphaPass; bool bDisableFog; // Utility @@ -112,6 +112,7 @@ struct VideoConfig final bool bEFBAccessEnable; bool bPerfQueriesEnable; bool bBBoxEnable; + bool bForceProgressive; bool bEFBEmulateFormatChanges; bool bSkipEFBCopyToRam; @@ -143,7 +144,7 @@ struct VideoConfig final API_TYPE APIType; std::vector<std::string> Adapters; // for D3D - std::vector<std::string> AAModes; + std::vector<int> AAModes; std::vector<std::string> PPShaders; // post-processing shaders std::vector<std::string> AnaglyphShaders; // anaglyph shaders @@ -160,7 +161,7 @@ struct VideoConfig final bool bSupportsPostProcessing; bool bSupportsPaletteConversion; bool bSupportsClipControl; // Needed by VertexShaderGen, so must stay in VideoCommon - bool bSupportsCopySubImage; // Needed for partial texture updates + bool bSupportsSSAA; } backend_info; // Utility diff --git a/Source/Core/VideoCommon/XFMemory.cpp b/Source/Core/VideoCommon/XFMemory.cpp index 0e54fea373..1e9e71c5b8 100644 --- a/Source/Core/VideoCommon/XFMemory.cpp +++ b/Source/Core/VideoCommon/XFMemory.cpp @@ -2,6 +2,7 @@ // Licensed under GPLv2+ // Refer to the license.txt file included. +#include "Common/Common.h" #include "VideoCommon/XFMemory.h" // STATE_TO_SAVE diff --git a/Source/Core/VideoCommon/XFMemory.h b/Source/Core/VideoCommon/XFMemory.h index 29a3aa9a65..86a1ce89a1 100644 --- a/Source/Core/VideoCommon/XFMemory.h +++ b/Source/Core/VideoCommon/XFMemory.h @@ -8,92 +8,126 @@ #include "VideoCommon/CPMemory.h" #include "VideoCommon/DataReader.h" - // Lighting +// Projection +enum : u32 +{ + XF_TEXPROJ_ST = 0, + XF_TEXPROJ_STQ = 1 +}; -#define XF_TEXPROJ_ST 0 -#define XF_TEXPROJ_STQ 1 - -#define XF_TEXINPUT_AB11 0 -#define XF_TEXINPUT_ABC1 1 +// Input form +enum : u32 +{ + XF_TEXINPUT_AB11 = 0, + XF_TEXINPUT_ABC1 = 1 +}; -#define XF_TEXGEN_REGULAR 0 -#define XF_TEXGEN_EMBOSS_MAP 1 // used when bump mapping -#define XF_TEXGEN_COLOR_STRGBC0 2 -#define XF_TEXGEN_COLOR_STRGBC1 3 +// Texture generation type +enum : u32 +{ + XF_TEXGEN_REGULAR = 0, + XF_TEXGEN_EMBOSS_MAP = 1, // Used when bump mapping + XF_TEXGEN_COLOR_STRGBC0 = 2, + XF_TEXGEN_COLOR_STRGBC1 = 3 +}; -#define XF_SRCGEOM_INROW 0 // input is abc -#define XF_SRCNORMAL_INROW 1 // input is abc -#define XF_SRCCOLORS_INROW 2 -#define XF_SRCBINORMAL_T_INROW 3 // input is abc -#define XF_SRCBINORMAL_B_INROW 4 // input is abc -#define XF_SRCTEX0_INROW 5 -#define XF_SRCTEX1_INROW 6 -#define XF_SRCTEX2_INROW 7 -#define XF_SRCTEX3_INROW 8 -#define XF_SRCTEX4_INROW 9 -#define XF_SRCTEX5_INROW 10 -#define XF_SRCTEX6_INROW 11 -#define XF_SRCTEX7_INROW 12 +// Source row +enum : u32 +{ + XF_SRCGEOM_INROW = 0, // Input is abc + XF_SRCNORMAL_INROW = 1, // Input is abc + XF_SRCCOLORS_INROW = 2, + XF_SRCBINORMAL_T_INROW = 3, // Input is abc + XF_SRCBINORMAL_B_INROW = 4, // Input is abc + XF_SRCTEX0_INROW = 5, + XF_SRCTEX1_INROW = 6, + XF_SRCTEX2_INROW = 7, + XF_SRCTEX3_INROW = 8, + XF_SRCTEX4_INROW = 9, + XF_SRCTEX5_INROW = 10, + XF_SRCTEX6_INROW = 11, + XF_SRCTEX7_INROW = 12 +}; -#define GX_SRC_REG 0 -#define GX_SRC_VTX 1 +// Control source +enum : u32 +{ + GX_SRC_REG = 0, + GX_SRC_VTX = 1 +}; -#define LIGHTDIF_NONE 0 -#define LIGHTDIF_SIGN 1 -#define LIGHTDIF_CLAMP 2 +// Light diffuse attenuation function +enum : u32 +{ + LIGHTDIF_NONE = 0, + LIGHTDIF_SIGN = 1, + LIGHTDIF_CLAMP = 2 +}; -#define LIGHTATTN_NONE 0 // no attenuation -#define LIGHTATTN_SPEC 1 // point light attenuation -#define LIGHTATTN_DIR 2 // directional light attenuation -#define LIGHTATTN_SPOT 3 // spot light attenuation +// Light attenuation function +enum : u32 +{ + LIGHTATTN_NONE = 0, // No attenuation + LIGHTATTN_SPEC = 1, // Point light attenuation + LIGHTATTN_DIR = 2, // Directional light attenuation + LIGHTATTN_SPOT = 3 // Spot light attenuation +}; -#define GX_PERSPECTIVE 0 -#define GX_ORTHOGRAPHIC 1 +// Projection type +enum : u32 +{ + GX_PERSPECTIVE = 0, + GX_ORTHOGRAPHIC = 1 +}; -#define XFMEM_POSMATRICES 0x000 -#define XFMEM_POSMATRICES_END 0x100 -#define XFMEM_NORMALMATRICES 0x400 -#define XFMEM_NORMALMATRICES_END 0x460 -#define XFMEM_POSTMATRICES 0x500 -#define XFMEM_POSTMATRICES_END 0x600 -#define XFMEM_LIGHTS 0x600 -#define XFMEM_LIGHTS_END 0x680 -#define XFMEM_ERROR 0x1000 -#define XFMEM_DIAG 0x1001 -#define XFMEM_STATE0 0x1002 -#define XFMEM_STATE1 0x1003 -#define XFMEM_CLOCK 0x1004 -#define XFMEM_CLIPDISABLE 0x1005 -#define XFMEM_SETGPMETRIC 0x1006 -#define XFMEM_VTXSPECS 0x1008 -#define XFMEM_SETNUMCHAN 0x1009 -#define XFMEM_SETCHAN0_AMBCOLOR 0x100a -#define XFMEM_SETCHAN1_AMBCOLOR 0x100b -#define XFMEM_SETCHAN0_MATCOLOR 0x100c -#define XFMEM_SETCHAN1_MATCOLOR 0x100d -#define XFMEM_SETCHAN0_COLOR 0x100e -#define XFMEM_SETCHAN1_COLOR 0x100f -#define XFMEM_SETCHAN0_ALPHA 0x1010 -#define XFMEM_SETCHAN1_ALPHA 0x1011 -#define XFMEM_DUALTEX 0x1012 -#define XFMEM_SETMATRIXINDA 0x1018 -#define XFMEM_SETMATRIXINDB 0x1019 -#define XFMEM_SETVIEWPORT 0x101a -#define XFMEM_SETZSCALE 0x101c -#define XFMEM_SETZOFFSET 0x101f -#define XFMEM_SETPROJECTION 0x1020 -/*#define XFMEM_SETPROJECTIONB 0x1021 -#define XFMEM_SETPROJECTIONC 0x1022 -#define XFMEM_SETPROJECTIOND 0x1023 -#define XFMEM_SETPROJECTIONE 0x1024 -#define XFMEM_SETPROJECTIONF 0x1025 -#define XFMEM_SETPROJECTIONORTHO1 0x1026 -#define XFMEM_SETPROJECTIONORTHO2 0x1027*/ -#define XFMEM_SETNUMTEXGENS 0x103f -#define XFMEM_SETTEXMTXINFO 0x1040 -#define XFMEM_SETPOSMTXINFO 0x1050 +// Registers and register ranges +enum +{ + XFMEM_POSMATRICES = 0x000, + XFMEM_POSMATRICES_END = 0x100, + XFMEM_NORMALMATRICES = 0x400, + XFMEM_NORMALMATRICES_END = 0x460, + XFMEM_POSTMATRICES = 0x500, + XFMEM_POSTMATRICES_END = 0x600, + XFMEM_LIGHTS = 0x600, + XFMEM_LIGHTS_END = 0x680, + XFMEM_ERROR = 0x1000, + XFMEM_DIAG = 0x1001, + XFMEM_STATE0 = 0x1002, + XFMEM_STATE1 = 0x1003, + XFMEM_CLOCK = 0x1004, + XFMEM_CLIPDISABLE = 0x1005, + XFMEM_SETGPMETRIC = 0x1006, + XFMEM_VTXSPECS = 0x1008, + XFMEM_SETNUMCHAN = 0x1009, + XFMEM_SETCHAN0_AMBCOLOR = 0x100a, + XFMEM_SETCHAN1_AMBCOLOR = 0x100b, + XFMEM_SETCHAN0_MATCOLOR = 0x100c, + XFMEM_SETCHAN1_MATCOLOR = 0x100d, + XFMEM_SETCHAN0_COLOR = 0x100e, + XFMEM_SETCHAN1_COLOR = 0x100f, + XFMEM_SETCHAN0_ALPHA = 0x1010, + XFMEM_SETCHAN1_ALPHA = 0x1011, + XFMEM_DUALTEX = 0x1012, + XFMEM_SETMATRIXINDA = 0x1018, + XFMEM_SETMATRIXINDB = 0x1019, + XFMEM_SETVIEWPORT = 0x101a, + XFMEM_SETZSCALE = 0x101c, + XFMEM_SETZOFFSET = 0x101f, + XFMEM_SETPROJECTION = 0x1020, + // XFMEM_SETPROJECTIONB = 0x1021, + // XFMEM_SETPROJECTIONC = 0x1022, + // XFMEM_SETPROJECTIOND = 0x1023, + // XFMEM_SETPROJECTIONE = 0x1024, + // XFMEM_SETPROJECTIONF = 0x1025, + // XFMEM_SETPROJECTIONORTHO1 = 0x1026, + // XFMEM_SETPROJECTIONORTHO2 = 0x1027, + XFMEM_SETNUMTEXGENS = 0x103f, + XFMEM_SETTEXMTXINFO = 0x1040, + XFMEM_SETPOSMTXINFO = 0x1050, +}; union LitChannel { diff --git a/Source/Core/VideoCommon/XFStructs.cpp b/Source/Core/VideoCommon/XFStructs.cpp index 089a458683..6a517dc0ab 100644 --- a/Source/Core/VideoCommon/XFStructs.cpp +++ b/Source/Core/VideoCommon/XFStructs.cpp @@ -62,7 +62,7 @@ static void XFRegWritten(int transferSize, u32 baseAddress, DataReader src) if (xfmem.ambColor[chan] != newValue) { VertexManager::Flush(); - VertexShaderManager::SetMaterialColorChanged(chan, newValue); + VertexShaderManager::SetMaterialColorChanged(chan); } break; } @@ -74,7 +74,7 @@ static void XFRegWritten(int transferSize, u32 baseAddress, DataReader src) if (xfmem.matColor[chan] != newValue) { VertexManager::Flush(); - VertexShaderManager::SetMaterialColorChanged(chan + 2, newValue); + VertexShaderManager::SetMaterialColorChanged(chan + 2); } break; } |
