summaryrefslogtreecommitdiff
path: root/Source/Core/VideoCommon
diff options
context:
space:
mode:
authorLinktothepast <kostamarino@gmail.com>2015-09-28 16:50:35 +0300
committerLinktothepast <kostamarino@gmail.com>2015-09-28 16:50:35 +0300
commit748be565b8df535ef558e74b38cd0f4d615f8a21 (patch)
treeb151efee0a87e99125f2401325a9bddecf84fc73 /Source/Core/VideoCommon
parentcf17fccdd3093a1b8476b10183b1c4bde36da74d (diff)
parentce493b897d6d3735c930a8465cc0c26bbe5feb86 (diff)
Merge remote-tracking branch 'dolphin-emu/master' into gameinis
Diffstat (limited to 'Source/Core/VideoCommon')
-rw-r--r--Source/Core/VideoCommon/AVIDump.cpp4
-rw-r--r--Source/Core/VideoCommon/BPFunctions.cpp10
-rw-r--r--Source/Core/VideoCommon/BPFunctions.h3
-rw-r--r--Source/Core/VideoCommon/BPMemory.h436
-rw-r--r--Source/Core/VideoCommon/BPStructs.cpp43
-rw-r--r--Source/Core/VideoCommon/BoundingBox.cpp10
-rw-r--r--Source/Core/VideoCommon/CommandProcessor.h2
-rw-r--r--Source/Core/VideoCommon/DataReader.h12
-rw-r--r--Source/Core/VideoCommon/Debugger.cpp5
-rw-r--r--Source/Core/VideoCommon/DriverDetails.cpp14
-rw-r--r--Source/Core/VideoCommon/DriverDetails.h108
-rw-r--r--Source/Core/VideoCommon/Fifo.cpp4
-rw-r--r--Source/Core/VideoCommon/FramebufferManagerBase.cpp29
-rw-r--r--Source/Core/VideoCommon/FramebufferManagerBase.h4
-rw-r--r--Source/Core/VideoCommon/GeometryShaderGen.cpp4
-rw-r--r--Source/Core/VideoCommon/HiresTextures.cpp5
-rw-r--r--Source/Core/VideoCommon/ImageWrite.cpp2
-rw-r--r--Source/Core/VideoCommon/IndexGenerator.cpp1
-rw-r--r--Source/Core/VideoCommon/LightingShaderGen.h5
-rw-r--r--Source/Core/VideoCommon/LookUpTables.h8
-rw-r--r--Source/Core/VideoCommon/NativeVertexFormat.h4
-rw-r--r--Source/Core/VideoCommon/OnScreenDisplay.cpp2
-rw-r--r--Source/Core/VideoCommon/OpcodeDecoding.h2
-rw-r--r--Source/Core/VideoCommon/PixelShaderGen.cpp213
-rw-r--r--Source/Core/VideoCommon/PixelShaderGen.h23
-rw-r--r--Source/Core/VideoCommon/PixelShaderManager.h1
-rw-r--r--Source/Core/VideoCommon/RenderBase.cpp85
-rw-r--r--Source/Core/VideoCommon/RenderBase.h2
-rw-r--r--Source/Core/VideoCommon/ShaderGenCommon.h23
-rw-r--r--Source/Core/VideoCommon/TextureCacheBase.cpp369
-rw-r--r--Source/Core/VideoCommon/TextureCacheBase.h36
-rw-r--r--Source/Core/VideoCommon/TextureConversionShader.cpp44
-rw-r--r--Source/Core/VideoCommon/TextureDecoder.h25
-rw-r--r--Source/Core/VideoCommon/TextureDecoder_Common.cpp23
-rw-r--r--Source/Core/VideoCommon/TextureDecoder_x64.cpp12
-rw-r--r--Source/Core/VideoCommon/VertexLoader.cpp1
-rw-r--r--Source/Core/VideoCommon/VertexLoaderARM64.cpp62
-rw-r--r--Source/Core/VideoCommon/VertexLoaderBase.cpp13
-rw-r--r--Source/Core/VideoCommon/VertexLoaderManager.cpp3
-rw-r--r--Source/Core/VideoCommon/VertexLoaderUtils.h19
-rw-r--r--Source/Core/VideoCommon/VertexLoaderX64.cpp19
-rw-r--r--Source/Core/VideoCommon/VertexLoader_Color.cpp80
-rw-r--r--Source/Core/VideoCommon/VertexLoader_Normal.cpp1
-rw-r--r--Source/Core/VideoCommon/VertexLoader_Position.cpp1
-rw-r--r--Source/Core/VideoCommon/VertexLoader_Position.h7
-rw-r--r--Source/Core/VideoCommon/VertexLoader_TextCoord.cpp1
-rw-r--r--Source/Core/VideoCommon/VertexLoader_TextCoord.h4
-rw-r--r--Source/Core/VideoCommon/VertexManagerBase.cpp3
-rw-r--r--Source/Core/VideoCommon/VertexShaderGen.cpp63
-rw-r--r--Source/Core/VideoCommon/VertexShaderManager.cpp33
-rw-r--r--Source/Core/VideoCommon/VertexShaderManager.h2
-rw-r--r--Source/Core/VideoCommon/VideoBackendBase.cpp18
-rw-r--r--Source/Core/VideoCommon/VideoBackendBase.h6
-rw-r--r--Source/Core/VideoCommon/VideoCommon.h14
-rw-r--r--Source/Core/VideoCommon/VideoCommon.vcxproj6
-rw-r--r--Source/Core/VideoCommon/VideoConfig.cpp10
-rw-r--r--Source/Core/VideoCommon/VideoConfig.h15
-rw-r--r--Source/Core/VideoCommon/XFMemory.cpp1
-rw-r--r--Source/Core/VideoCommon/XFMemory.h186
-rw-r--r--Source/Core/VideoCommon/XFStructs.cpp4
60 files changed, 1168 insertions, 977 deletions
diff --git a/Source/Core/VideoCommon/AVIDump.cpp b/Source/Core/VideoCommon/AVIDump.cpp
index fe5b0050e7..0424082709 100644
--- a/Source/Core/VideoCommon/AVIDump.cpp
+++ b/Source/Core/VideoCommon/AVIDump.cpp
@@ -480,12 +480,12 @@ void AVIDump::AddFrame(const u8* data, int width, int height)
while (!error && got_packet)
{
// Write the compressed frame in the media file.
- if (pkt.pts != AV_NOPTS_VALUE)
+ if (pkt.pts != (s64)AV_NOPTS_VALUE)
{
pkt.pts = av_rescale_q(pkt.pts,
s_stream->codec->time_base, s_stream->time_base);
}
- if (pkt.dts != AV_NOPTS_VALUE)
+ if (pkt.dts != (s64)AV_NOPTS_VALUE)
{
pkt.dts = av_rescale_q(pkt.dts,
s_stream->codec->time_base, s_stream->time_base);
diff --git a/Source/Core/VideoCommon/BPFunctions.cpp b/Source/Core/VideoCommon/BPFunctions.cpp
index 4150a7cf31..e4b5f54901 100644
--- a/Source/Core/VideoCommon/BPFunctions.cpp
+++ b/Source/Core/VideoCommon/BPFunctions.cpp
@@ -9,7 +9,6 @@
#include "VideoCommon/BPFunctions.h"
#include "VideoCommon/RenderBase.h"
-#include "VideoCommon/TextureCacheBase.h"
#include "VideoCommon/VertexManagerBase.h"
#include "VideoCommon/VertexShaderManager.h"
#include "VideoCommon/VideoConfig.h"
@@ -85,15 +84,6 @@ void SetColorMask()
g_renderer->SetColorMask();
}
-void CopyEFB(u32 dstAddr, const EFBRectangle& srcRect,
- unsigned int dstFormat, PEControl::PixelFormat srcFormat,
- bool isIntensity, bool scaleByHalf)
-{
- // bpmem.zcontrol.pixel_format to PEControl::Z24 is when the game wants to copy from ZBuffer (Zbuffer uses 24-bit Format)
- TextureCache::CopyRenderTargetToTexture(dstAddr, dstFormat, srcFormat,
- srcRect, isIntensity, scaleByHalf);
-}
-
/* Explanation of the magic behind ClearScreen:
There's numerous possible formats for the pixel data in the EFB.
However, in the HW accelerated backends we're always using RGBA8
diff --git a/Source/Core/VideoCommon/BPFunctions.h b/Source/Core/VideoCommon/BPFunctions.h
index 890f734deb..126d9f1d7b 100644
--- a/Source/Core/VideoCommon/BPFunctions.h
+++ b/Source/Core/VideoCommon/BPFunctions.h
@@ -23,9 +23,6 @@ void SetBlendMode();
void SetDitherMode();
void SetLogicOpMode();
void SetColorMask();
-void CopyEFB(u32 dstAddr, const EFBRectangle& srcRect,
- unsigned int dstFormat, PEControl::PixelFormat srcFormat,
- bool isIntensity, bool scaleByHalf);
void ClearScreen(const EFBRectangle &rc);
void OnPixelFormatChange();
void SetInterlacingMode(const BPCmd &bp);
diff --git a/Source/Core/VideoCommon/BPMemory.h b/Source/Core/VideoCommon/BPMemory.h
index 50a95791ce..14ac732edf 100644
--- a/Source/Core/VideoCommon/BPMemory.h
+++ b/Source/Core/VideoCommon/BPMemory.h
@@ -11,154 +11,239 @@
#pragma pack(4)
-#define BPMEM_GENMODE 0x00
-#define BPMEM_DISPLAYCOPYFILTER 0x01 // 0x01 + 4
-#define BPMEM_IND_MTXA 0x06 // 0x06 + (3 * 3)
-#define BPMEM_IND_MTXB 0x07 // 0x07 + (3 * 3)
-#define BPMEM_IND_MTXC 0x08 // 0x08 + (3 * 3)
-#define BPMEM_IND_IMASK 0x0F
-#define BPMEM_IND_CMD 0x10 // 0x10 + 16
-#define BPMEM_SCISSORTL 0x20
-#define BPMEM_SCISSORBR 0x21
-#define BPMEM_LINEPTWIDTH 0x22
-#define BPMEM_PERF0_TRI 0x23
-#define BPMEM_PERF0_QUAD 0x24
-#define BPMEM_RAS1_SS0 0x25
-#define BPMEM_RAS1_SS1 0x26
-#define BPMEM_IREF 0x27
-#define BPMEM_TREF 0x28 // 0x28 + 8
-#define BPMEM_SU_SSIZE 0x30 // 0x30 + (2 * 8)
-#define BPMEM_SU_TSIZE 0x31 // 0x31 + (2 * 8)
-#define BPMEM_ZMODE 0x40
-#define BPMEM_BLENDMODE 0x41
-#define BPMEM_CONSTANTALPHA 0x42
-#define BPMEM_ZCOMPARE 0x43
-#define BPMEM_FIELDMASK 0x44
-#define BPMEM_SETDRAWDONE 0x45
-#define BPMEM_BUSCLOCK0 0x46
-#define BPMEM_PE_TOKEN_ID 0x47
-#define BPMEM_PE_TOKEN_INT_ID 0x48
-#define BPMEM_EFB_TL 0x49
-#define BPMEM_EFB_BR 0x4A
-#define BPMEM_EFB_ADDR 0x4B
-#define BPMEM_MIPMAP_STRIDE 0x4D
-#define BPMEM_COPYYSCALE 0x4E
-#define BPMEM_CLEAR_AR 0x4F
-#define BPMEM_CLEAR_GB 0x50
-#define BPMEM_CLEAR_Z 0x51
-#define BPMEM_TRIGGER_EFB_COPY 0x52
-#define BPMEM_COPYFILTER0 0x53
-#define BPMEM_COPYFILTER1 0x54
-#define BPMEM_CLEARBBOX1 0x55
-#define BPMEM_CLEARBBOX2 0x56
-#define BPMEM_CLEAR_PIXEL_PERF 0x57
-#define BPMEM_REVBITS 0x58
-#define BPMEM_SCISSOROFFSET 0x59
-#define BPMEM_PRELOAD_ADDR 0x60
-#define BPMEM_PRELOAD_TMEMEVEN 0x61
-#define BPMEM_PRELOAD_TMEMODD 0x62
-#define BPMEM_PRELOAD_MODE 0x63
-#define BPMEM_LOADTLUT0 0x64
-#define BPMEM_LOADTLUT1 0x65
-#define BPMEM_TEXINVALIDATE 0x66
-#define BPMEM_PERF1 0x67
-#define BPMEM_FIELDMODE 0x68
-#define BPMEM_BUSCLOCK1 0x69
-#define BPMEM_TX_SETMODE0 0x80 // 0x80 + 4
-#define BPMEM_TX_SETMODE1 0x84 // 0x84 + 4
-#define BPMEM_TX_SETIMAGE0 0x88 // 0x88 + 4
-#define BPMEM_TX_SETIMAGE1 0x8C // 0x8C + 4
-#define BPMEM_TX_SETIMAGE2 0x90 // 0x90 + 4
-#define BPMEM_TX_SETIMAGE3 0x94 // 0x94 + 4
-#define BPMEM_TX_SETTLUT 0x98 // 0x98 + 4
-#define BPMEM_TX_SETMODE0_4 0xA0 // 0xA0 + 4
-#define BPMEM_TX_SETMODE1_4 0xA4 // 0xA4 + 4
-#define BPMEM_TX_SETIMAGE0_4 0xA8 // 0xA8 + 4
-#define BPMEM_TX_SETIMAGE1_4 0xAC // 0xA4 + 4
-#define BPMEM_TX_SETIMAGE2_4 0xB0 // 0xB0 + 4
-#define BPMEM_TX_SETIMAGE3_4 0xB4 // 0xB4 + 4
-#define BPMEM_TX_SETTLUT_4 0xB8 // 0xB8 + 4
-#define BPMEM_TEV_COLOR_ENV 0xC0 // 0xC0 + (2 * 16)
-#define BPMEM_TEV_ALPHA_ENV 0xC1 // 0xC1 + (2 * 16)
-#define BPMEM_TEV_COLOR_RA 0xE0 // 0xE0 + (2 * 4)
-#define BPMEM_TEV_COLOR_BG 0xE1 // 0xE1 + (2 * 4)
-#define BPMEM_FOGRANGE 0xE8 // 0xE8 + 6
-#define BPMEM_FOGPARAM0 0xEE
-#define BPMEM_FOGBMAGNITUDE 0xEF
-#define BPMEM_FOGBEXPONENT 0xF0
-#define BPMEM_FOGPARAM3 0xF1
-#define BPMEM_FOGCOLOR 0xF2
-#define BPMEM_ALPHACOMPARE 0xF3
-#define BPMEM_BIAS 0xF4
-#define BPMEM_ZTEX2 0xF5
-#define BPMEM_TEV_KSEL 0xF6 // 0xF6 + 8
-#define BPMEM_BP_MASK 0xFE
+enum
+{
+ BPMEM_GENMODE = 0x00,
+ BPMEM_DISPLAYCOPYFILTER = 0x01, // 0x01 + 4
+ BPMEM_IND_MTXA = 0x06, // 0x06 + (3 * 3)
+ BPMEM_IND_MTXB = 0x07, // 0x07 + (3 * 3)
+ BPMEM_IND_MTXC = 0x08, // 0x08 + (3 * 3)
+ BPMEM_IND_IMASK = 0x0F,
+ BPMEM_IND_CMD = 0x10, // 0x10 + 16
+ BPMEM_SCISSORTL = 0x20,
+ BPMEM_SCISSORBR = 0x21,
+ BPMEM_LINEPTWIDTH = 0x22,
+ BPMEM_PERF0_TRI = 0x23,
+ BPMEM_PERF0_QUAD = 0x24,
+ BPMEM_RAS1_SS0 = 0x25,
+ BPMEM_RAS1_SS1 = 0x26,
+ BPMEM_IREF = 0x27,
+ BPMEM_TREF = 0x28, // 0x28 + 8
+ BPMEM_SU_SSIZE = 0x30, // 0x30 + (2 * 8)
+ BPMEM_SU_TSIZE = 0x31, // 0x31 + (2 * 8)
+ BPMEM_ZMODE = 0x40,
+ BPMEM_BLENDMODE = 0x41,
+ BPMEM_CONSTANTALPHA = 0x42,
+ BPMEM_ZCOMPARE = 0x43,
+ BPMEM_FIELDMASK = 0x44,
+ BPMEM_SETDRAWDONE = 0x45,
+ BPMEM_BUSCLOCK0 = 0x46,
+ BPMEM_PE_TOKEN_ID = 0x47,
+ BPMEM_PE_TOKEN_INT_ID = 0x48,
+ BPMEM_EFB_TL = 0x49,
+ BPMEM_EFB_BR = 0x4A,
+ BPMEM_EFB_ADDR = 0x4B,
+ BPMEM_MIPMAP_STRIDE = 0x4D,
+ BPMEM_COPYYSCALE = 0x4E,
+ BPMEM_CLEAR_AR = 0x4F,
+ BPMEM_CLEAR_GB = 0x50,
+ BPMEM_CLEAR_Z = 0x51,
+ BPMEM_TRIGGER_EFB_COPY = 0x52,
+ BPMEM_COPYFILTER0 = 0x53,
+ BPMEM_COPYFILTER1 = 0x54,
+ BPMEM_CLEARBBOX1 = 0x55,
+ BPMEM_CLEARBBOX2 = 0x56,
+ BPMEM_CLEAR_PIXEL_PERF = 0x57,
+ BPMEM_REVBITS = 0x58,
+ BPMEM_SCISSOROFFSET = 0x59,
+ BPMEM_PRELOAD_ADDR = 0x60,
+ BPMEM_PRELOAD_TMEMEVEN = 0x61,
+ BPMEM_PRELOAD_TMEMODD = 0x62,
+ BPMEM_PRELOAD_MODE = 0x63,
+ BPMEM_LOADTLUT0 = 0x64,
+ BPMEM_LOADTLUT1 = 0x65,
+ BPMEM_TEXINVALIDATE = 0x66,
+ BPMEM_PERF1 = 0x67,
+ BPMEM_FIELDMODE = 0x68,
+ BPMEM_BUSCLOCK1 = 0x69,
+ BPMEM_TX_SETMODE0 = 0x80, // 0x80 + 4
+ BPMEM_TX_SETMODE1 = 0x84, // 0x84 + 4
+ BPMEM_TX_SETIMAGE0 = 0x88, // 0x88 + 4
+ BPMEM_TX_SETIMAGE1 = 0x8C, // 0x8C + 4
+ BPMEM_TX_SETIMAGE2 = 0x90, // 0x90 + 4
+ BPMEM_TX_SETIMAGE3 = 0x94, // 0x94 + 4
+ BPMEM_TX_SETTLUT = 0x98, // 0x98 + 4
+ BPMEM_TX_SETMODE0_4 = 0xA0, // 0xA0 + 4
+ BPMEM_TX_SETMODE1_4 = 0xA4, // 0xA4 + 4
+ BPMEM_TX_SETIMAGE0_4 = 0xA8, // 0xA8 + 4
+ BPMEM_TX_SETIMAGE1_4 = 0xAC, // 0xA4 + 4
+ BPMEM_TX_SETIMAGE2_4 = 0xB0, // 0xB0 + 4
+ BPMEM_TX_SETIMAGE3_4 = 0xB4, // 0xB4 + 4
+ BPMEM_TX_SETTLUT_4 = 0xB8, // 0xB8 + 4
+ BPMEM_TEV_COLOR_ENV = 0xC0, // 0xC0 + (2 * 16)
+ BPMEM_TEV_ALPHA_ENV = 0xC1, // 0xC1 + (2 * 16)
+ BPMEM_TEV_COLOR_RA = 0xE0, // 0xE0 + (2 * 4)
+ BPMEM_TEV_COLOR_BG = 0xE1, // 0xE1 + (2 * 4)
+ BPMEM_FOGRANGE = 0xE8, // 0xE8 + 6
+ BPMEM_FOGPARAM0 = 0xEE,
+ BPMEM_FOGBMAGNITUDE = 0xEF,
+ BPMEM_FOGBEXPONENT = 0xF0,
+ BPMEM_FOGPARAM3 = 0xF1,
+ BPMEM_FOGCOLOR = 0xF2,
+ BPMEM_ALPHACOMPARE = 0xF3,
+ BPMEM_BIAS = 0xF4,
+ BPMEM_ZTEX2 = 0xF5,
+ BPMEM_TEV_KSEL = 0xF6, // 0xF6 + 8
+ BPMEM_BP_MASK = 0xFE,
+};
// Tev/combiner things
-#define TEVSCALE_1 0
-#define TEVSCALE_2 1
-#define TEVSCALE_4 2
-#define TEVDIVIDE_2 3
-
-#define TEVCMP_R8 0
-#define TEVCMP_GR16 1
-#define TEVCMP_BGR24 2
-#define TEVCMP_RGB8 3
-
-#define TEVOP_ADD 0
-#define TEVOP_SUB 1
-#define TEVCMP_R8_GT 8
-#define TEVCMP_R8_EQ 9
-#define TEVCMP_GR16_GT 10
-#define TEVCMP_GR16_EQ 11
-#define TEVCMP_BGR24_GT 12
-#define TEVCMP_BGR24_EQ 13
-#define TEVCMP_RGB8_GT 14
-#define TEVCMP_RGB8_EQ 15
-#define TEVCMP_A8_GT 14
-#define TEVCMP_A8_EQ 15
-
-#define TEVCOLORARG_CPREV 0
-#define TEVCOLORARG_APREV 1
-#define TEVCOLORARG_C0 2
-#define TEVCOLORARG_A0 3
-#define TEVCOLORARG_C1 4
-#define TEVCOLORARG_A1 5
-#define TEVCOLORARG_C2 6
-#define TEVCOLORARG_A2 7
-#define TEVCOLORARG_TEXC 8
-#define TEVCOLORARG_TEXA 9
-#define TEVCOLORARG_RASC 10
-#define TEVCOLORARG_RASA 11
-#define TEVCOLORARG_ONE 12
-#define TEVCOLORARG_HALF 13
-#define TEVCOLORARG_KONST 14
-#define TEVCOLORARG_ZERO 15
-
-#define TEVALPHAARG_APREV 0
-#define TEVALPHAARG_A0 1
-#define TEVALPHAARG_A1 2
-#define TEVALPHAARG_A2 3
-#define TEVALPHAARG_TEXA 4
-#define TEVALPHAARG_RASA 5
-#define TEVALPHAARG_KONST 6
-#define TEVALPHAARG_ZERO 7
-
-#define GX_TEVPREV 0
-#define GX_TEVREG0 1
-#define GX_TEVREG1 2
-#define GX_TEVREG2 3
-
-#define ZTEXTURE_DISABLE 0
-#define ZTEXTURE_ADD 1
-#define ZTEXTURE_REPLACE 2
-
-#define TevBias_ZERO 0
-#define TevBias_ADDHALF 1
-#define TevBias_SUBHALF 2
-#define TevBias_COMPARE 3
+// TEV scaling type
+enum : u32
+{
+ TEVSCALE_1 = 0,
+ TEVSCALE_2 = 1,
+ TEVSCALE_4 = 2,
+ TEVDIVIDE_2 = 3
+};
+
+enum : u32
+{
+ TEVCMP_R8 = 0,
+ TEVCMP_GR16 = 1,
+ TEVCMP_BGR24 = 2,
+ TEVCMP_RGB8 = 3
+};
+
+// TEV combiner operator
+enum : u32
+{
+ TEVOP_ADD = 0,
+ TEVOP_SUB = 1,
+ TEVCMP_R8_GT = 8,
+ TEVCMP_R8_EQ = 9,
+ TEVCMP_GR16_GT = 10,
+ TEVCMP_GR16_EQ = 11,
+ TEVCMP_BGR24_GT = 12,
+ TEVCMP_BGR24_EQ = 13,
+ TEVCMP_RGB8_GT = 14,
+ TEVCMP_RGB8_EQ = 15,
+ TEVCMP_A8_GT = TEVCMP_RGB8_GT,
+ TEVCMP_A8_EQ = TEVCMP_RGB8_EQ
+};
+
+// TEV color combiner input
+enum : u32
+{
+ TEVCOLORARG_CPREV = 0,
+ TEVCOLORARG_APREV = 1,
+ TEVCOLORARG_C0 = 2,
+ TEVCOLORARG_A0 = 3,
+ TEVCOLORARG_C1 = 4,
+ TEVCOLORARG_A1 = 5,
+ TEVCOLORARG_C2 = 6,
+ TEVCOLORARG_A2 = 7,
+ TEVCOLORARG_TEXC = 8,
+ TEVCOLORARG_TEXA = 9,
+ TEVCOLORARG_RASC = 10,
+ TEVCOLORARG_RASA = 11,
+ TEVCOLORARG_ONE = 12,
+ TEVCOLORARG_HALF = 13,
+ TEVCOLORARG_KONST = 14,
+ TEVCOLORARG_ZERO = 15
+};
+
+// TEV alpha combiner input
+enum : u32
+{
+ TEVALPHAARG_APREV = 0,
+ TEVALPHAARG_A0 = 1,
+ TEVALPHAARG_A1 = 2,
+ TEVALPHAARG_A2 = 3,
+ TEVALPHAARG_TEXA = 4,
+ TEVALPHAARG_RASA = 5,
+ TEVALPHAARG_KONST = 6,
+ TEVALPHAARG_ZERO = 7
+};
+
+// TEV output registers
+enum : u32
+{
+ GX_TEVPREV = 0,
+ GX_TEVREG0 = 1,
+ GX_TEVREG1 = 2,
+ GX_TEVREG2 = 3
+};
+
+// Z-texture formats
+enum : u32
+{
+ TEV_ZTEX_TYPE_U8 = 0,
+ TEV_ZTEX_TYPE_U16 = 1,
+ TEV_ZTEX_TYPE_U24 = 2
+};
+
+// Z texture operator
+enum : u32
+{
+ ZTEXTURE_DISABLE = 0,
+ ZTEXTURE_ADD = 1,
+ ZTEXTURE_REPLACE = 2
+};
+
+// TEV bias value
+enum : u32
+{
+ TEVBIAS_ZERO = 0,
+ TEVBIAS_ADDHALF = 1,
+ TEVBIAS_SUBHALF = 2,
+ TEVBIAS_COMPARE = 3
+};
+
+// Indirect texture format
+enum : u32
+{
+ ITF_8 = 0,
+ ITF_5 = 1,
+ ITF_4 = 2,
+ ITF_3 = 3
+};
+
+// Indirect texture bias
+enum : u32
+{
+ ITB_NONE = 0,
+ ITB_S = 1,
+ ITB_T = 2,
+ ITB_ST = 3,
+ ITB_U = 4,
+ ITB_SU = 5,
+ ITB_TU = 6,
+ ITB_STU = 7
+};
+
+// Indirect texture bump alpha
+enum : u32
+{
+ ITBA_OFF = 0,
+ ITBA_S = 1,
+ ITBA_T = 2,
+ ITBA_U = 3
+};
+
+// Indirect texture wrap value
+enum : u32
+{
+ ITW_OFF = 0,
+ ITW_256 = 1,
+ ITW_128 = 2,
+ ITW_64 = 3,
+ ITW_32 = 4,
+ ITW_16 = 5,
+ ITW_0 = 6
+};
union IND_MTXA
{
@@ -213,32 +298,6 @@ union IND_IMASK
u32 hex;
};
-#define TEVSELCC_CPREV 0
-#define TEVSELCC_APREV 1
-#define TEVSELCC_C0 2
-#define TEVSELCC_A0 3
-#define TEVSELCC_C1 4
-#define TEVSELCC_A1 5
-#define TEVSELCC_C2 6
-#define TEVSELCC_A2 7
-#define TEVSELCC_TEXC 8
-#define TEVSELCC_TEXA 9
-#define TEVSELCC_RASC 10
-#define TEVSELCC_RASA 11
-#define TEVSELCC_ONE 12
-#define TEVSELCC_HALF 13
-#define TEVSELCC_KONST 14
-#define TEVSELCC_ZERO 15
-
-#define TEVSELCA_APREV 0
-#define TEVSELCA_A0 1
-#define TEVSELCA_A1 2
-#define TEVSELCA_A2 3
-#define TEVSELCA_TEXA 4
-#define TEVSELCA_RASA 5
-#define TEVSELCA_KONST 6
-#define TEVSELCA_ZERO 7
-
struct TevStageCombiner
{
union ColorCombiner
@@ -285,33 +344,6 @@ struct TevStageCombiner
AlphaCombiner alphaC;
};
-#define ITF_8 0
-#define ITF_5 1
-#define ITF_4 2
-#define ITF_3 3
-
-#define ITB_NONE 0
-#define ITB_S 1
-#define ITB_T 2
-#define ITB_ST 3
-#define ITB_U 4
-#define ITB_SU 5
-#define ITB_TU 6
-#define ITB_STU 7
-
-#define ITBA_OFF 0
-#define ITBA_S 1
-#define ITBA_T 2
-#define ITBA_U 3
-
-#define ITW_OFF 0
-#define ITW_256 1
-#define ITW_128 2
-#define ITW_64 3
-#define ITW_32 4
-#define ITW_16 5
-#define ITW_0 6
-
// several discoveries:
// GXSetTevIndBumpST(tevstage, indstage, matrixind)
// if ( matrix == 2 ) realmat = 6; // 10
@@ -513,16 +545,6 @@ union ZTex2
u32 hex;
};
-// Z-texture types (formats)
-#define TEV_ZTEX_TYPE_U8 0
-#define TEV_ZTEX_TYPE_U16 1
-#define TEV_ZTEX_TYPE_U24 2
-
-#define TEV_ZTEX_DISABLE 0
-#define TEV_ZTEX_ADD 1
-#define TEV_ZTEX_REPLACE 2
-
-
struct FourTexUnits
{
TexMode0 texMode0[4];
diff --git a/Source/Core/VideoCommon/BPStructs.cpp b/Source/Core/VideoCommon/BPStructs.cpp
index 410d4d44cc..4e3df7cc00 100644
--- a/Source/Core/VideoCommon/BPStructs.cpp
+++ b/Source/Core/VideoCommon/BPStructs.cpp
@@ -8,6 +8,7 @@
#include "Common/Thread.h"
#include "Core/ConfigManager.h"
#include "Core/Core.h"
+#include "Core/FifoPlayer/FifoRecorder.h"
#include "Core/HW/Memmap.h"
#include "VideoCommon/BoundingBox.h"
@@ -20,6 +21,7 @@
#include "VideoCommon/PixelShaderManager.h"
#include "VideoCommon/RenderBase.h"
#include "VideoCommon/Statistics.h"
+#include "VideoCommon/TextureCacheBase.h"
#include "VideoCommon/TextureDecoder.h"
#include "VideoCommon/VertexShaderManager.h"
#include "VideoCommon/VideoCommon.h"
@@ -205,6 +207,7 @@ static void BPWritten(const BPCmd& bp)
// The values in bpmem.copyTexSrcXY and bpmem.copyTexSrcWH are updated in case 0x49 and 0x4a in this function
u32 destAddr = bpmem.copyTexDest << 5;
+ u32 destStride = bpmem.copyMipMapStrideChannels << 5;
EFBRectangle srcRect;
srcRect.left = (int)bpmem.copyTexSrcXY.x;
@@ -223,8 +226,9 @@ static void BPWritten(const BPCmd& bp)
if (g_ActiveConfig.bShowEFBCopyRegions)
stats.efb_regions.push_back(srcRect);
- CopyEFB(destAddr, srcRect,
- PE_copy.tp_realFormat(), bpmem.zcontrol.pixel_format,
+ // bpmem.zcontrol.pixel_format to PEControl::Z24 is when the game wants to copy from ZBuffer (Zbuffer uses 24-bit Format)
+ TextureCache::CopyRenderTargetToTexture(destAddr, PE_copy.tp_realFormat(), destStride,
+ bpmem.zcontrol.pixel_format, srcRect,
!!PE_copy.intensity_fmt, !!PE_copy.half_scale);
}
else
@@ -241,7 +245,7 @@ static void BPWritten(const BPCmd& bp)
else
yScale = (float)bpmem.dispcopyyscale / 256.0f;
- float num_xfb_lines = ((bpmem.copyTexSrcWH.y + 1.0f) * yScale);
+ float num_xfb_lines = 1.0f + bpmem.copyTexSrcWH.y * yScale;
u32 height = static_cast<u32>(num_xfb_lines);
if (height > MAX_XFB_HEIGHT)
@@ -251,9 +255,9 @@ static void BPWritten(const BPCmd& bp)
height = MAX_XFB_HEIGHT;
}
- u32 width = bpmem.copyMipMapStrideChannels << 4;
-
- Renderer::RenderToXFB(destAddr, srcRect, width, height, s_gammaLUT[PE_copy.gamma]);
+ DEBUG_LOG(VIDEO, "RenderToXFB: destAddr: %08x | srcRect {%d %d %d %d} | fbWidth: %u | fbStride: %u | fbHeight: %u",
+ destAddr, srcRect.left, srcRect.top, srcRect.right, srcRect.bottom, bpmem.copyTexSrcWH.x + 1, destStride, height);
+ Renderer::RenderToXFB(destAddr, srcRect, destStride, height, s_gammaLUT[PE_copy.gamma]);
}
// Clear the rectangular region after copying it.
@@ -278,6 +282,9 @@ static void BPWritten(const BPCmd& bp)
Memory::CopyFromEmu(texMem + tlutTMemAddr, addr, tlutXferCount);
+ if (g_bRecordFifoData)
+ FifoRecorder::GetInstance().UseMemory(addr, tlutXferCount, MemoryUpdate::TMEM);
+
return;
}
case BPMEM_FOGRANGE: // Fog Settings Control
@@ -452,15 +459,16 @@ static void BPWritten(const BPCmd& bp)
BPS_TmemConfig& tmem_cfg = bpmem.tmem_config;
u32 src_addr = tmem_cfg.preload_addr << 5; // TODO: Should we add mask here on GC?
- u32 size = tmem_cfg.preload_tile_info.count * TMEM_LINE_SIZE;
+ u32 bytes_read = 0;
u32 tmem_addr_even = tmem_cfg.preload_tmem_even * TMEM_LINE_SIZE;
if (tmem_cfg.preload_tile_info.type != 3)
{
- if (tmem_addr_even + size > TMEM_SIZE)
- size = TMEM_SIZE - tmem_addr_even;
+ bytes_read = tmem_cfg.preload_tile_info.count * TMEM_LINE_SIZE;
+ if (tmem_addr_even + bytes_read > TMEM_SIZE)
+ bytes_read = TMEM_SIZE - tmem_addr_even;
- Memory::CopyFromEmu(texMem + tmem_addr_even, src_addr, size);
+ Memory::CopyFromEmu(texMem + tmem_addr_even, src_addr, bytes_read);
}
else // RGBA8 tiles (and CI14, but that might just be stupid libogc!)
{
@@ -471,18 +479,19 @@ static void BPWritten(const BPCmd& bp)
for (u32 i = 0; i < tmem_cfg.preload_tile_info.count; ++i)
{
- if (tmem_addr_even + TMEM_LINE_SIZE > TMEM_SIZE ||
- tmem_addr_odd + TMEM_LINE_SIZE > TMEM_SIZE)
- return;
+ if (tmem_addr_even + TMEM_LINE_SIZE > TMEM_SIZE || tmem_addr_odd + TMEM_LINE_SIZE > TMEM_SIZE)
+ break;
- // TODO: This isn't very optimised, does a whole lot of small memcpys
- memcpy(texMem + tmem_addr_even, src_ptr, TMEM_LINE_SIZE);
- memcpy(texMem + tmem_addr_odd, src_ptr + TMEM_LINE_SIZE, TMEM_LINE_SIZE);
+ memcpy(texMem + tmem_addr_even, src_ptr + bytes_read, TMEM_LINE_SIZE);
+ memcpy(texMem + tmem_addr_odd, src_ptr + bytes_read + TMEM_LINE_SIZE, TMEM_LINE_SIZE);
tmem_addr_even += TMEM_LINE_SIZE;
tmem_addr_odd += TMEM_LINE_SIZE;
- src_ptr += TMEM_LINE_SIZE * 2;
+ bytes_read += TMEM_LINE_SIZE * 2;
}
}
+
+ if (g_bRecordFifoData)
+ FifoRecorder::GetInstance().UseMemory(src_addr, bytes_read, MemoryUpdate::TMEM);
}
return;
diff --git a/Source/Core/VideoCommon/BoundingBox.cpp b/Source/Core/VideoCommon/BoundingBox.cpp
index f1bf5d066e..9336a6a671 100644
--- a/Source/Core/VideoCommon/BoundingBox.cpp
+++ b/Source/Core/VideoCommon/BoundingBox.cpp
@@ -2,15 +2,9 @@
// Licensed under GPLv2+
// Refer to the license.txt file included.
-
-
-#include "VideoBackends/Software/Clipper.h"
-#include "VideoBackends/Software/Rasterizer.h"
-#include "VideoBackends/Software/SetupUnit.h"
-#include "VideoBackends/Software/TransformUnit.h"
+#include "Common/ChunkFile.h"
+#include "Common/CommonTypes.h"
#include "VideoCommon/BoundingBox.h"
-#include "VideoCommon/PixelShaderManager.h"
-
namespace BoundingBox
{
diff --git a/Source/Core/VideoCommon/CommandProcessor.h b/Source/Core/VideoCommon/CommandProcessor.h
index d5061efd02..2618a19801 100644
--- a/Source/Core/VideoCommon/CommandProcessor.h
+++ b/Source/Core/VideoCommon/CommandProcessor.h
@@ -11,8 +11,6 @@
class PointerWrap;
namespace MMIO { class Mapping; }
-extern bool MT;
-
namespace CommandProcessor
{
diff --git a/Source/Core/VideoCommon/DataReader.h b/Source/Core/VideoCommon/DataReader.h
index 0737a9bf1d..bb4914db8c 100644
--- a/Source/Core/VideoCommon/DataReader.h
+++ b/Source/Core/VideoCommon/DataReader.h
@@ -4,7 +4,9 @@
#pragma once
-#include "Common/Common.h"
+#include <cstring>
+#include "Common/CommonFuncs.h"
+#include "Common/CommonTypes.h"
class DataReader
{
@@ -33,9 +35,12 @@ public:
template <typename T, bool swapped = true> __forceinline T Peek(int offset = 0)
{
- T data = *(T*)(buffer + offset);
+ T data;
+ std::memcpy(&data, &buffer[offset], sizeof(T));
+
if (swapped)
data = Common::FromBigEndian(data);
+
return data;
}
@@ -50,7 +55,8 @@ public:
{
if (swapped)
data = Common::FromBigEndian(data);
- *(T*)(buffer) = data;
+
+ std::memcpy(buffer, &data, sizeof(T));
buffer += sizeof(T);
}
diff --git a/Source/Core/VideoCommon/Debugger.cpp b/Source/Core/VideoCommon/Debugger.cpp
index 0c88be57f5..24ec679400 100644
--- a/Source/Core/VideoCommon/Debugger.cpp
+++ b/Source/Core/VideoCommon/Debugger.cpp
@@ -6,6 +6,7 @@
#include "Common/FileUtil.h"
#include "Common/IniFile.h"
+#include "Common/Thread.h"
#include "VideoCommon/Debugger.h"
#include "VideoCommon/NativeVertexFormat.h"
@@ -56,7 +57,7 @@ void GFXDebuggerCheckAndPause(bool update)
while ( GFXDebuggerPauseFlag )
{
if (update) GFXDebuggerUpdateScreen();
- SLEEP(5);
+ Common::SleepCurrentThread(5);
}
g_pdebugger->OnContinue();
}
@@ -81,7 +82,7 @@ void GFXDebuggerBase::DumpPixelShader(const std::string& path)
const std::string filename = StringFromFormat("%sdump_ps.txt", path.c_str());
std::string output;
- bool useDstAlpha = !g_ActiveConfig.bDstAlphaPass && bpmem.dstalpha.enable && bpmem.blendmode.alphaupdate && bpmem.zcontrol.pixel_format == PEControl::RGBA6_Z24;
+ bool useDstAlpha = bpmem.dstalpha.enable && bpmem.blendmode.alphaupdate && bpmem.zcontrol.pixel_format == PEControl::RGBA6_Z24;
if (!useDstAlpha)
{
output = "Destination alpha disabled:\n";
diff --git a/Source/Core/VideoCommon/DriverDetails.cpp b/Source/Core/VideoCommon/DriverDetails.cpp
index d3b1883cb3..cb9d01ba35 100644
--- a/Source/Core/VideoCommon/DriverDetails.cpp
+++ b/Source/Core/VideoCommon/DriverDetails.cpp
@@ -42,20 +42,16 @@ namespace DriverDetails
// This is a list of all known bugs for each vendor
// We use this to check if the device and driver has a issue
static BugInfo m_known_bugs[] = {
- {OS_ALL, VENDOR_QUALCOMM, DRIVER_QUALCOMM, -1, BUG_NODYNUBOACCESS, 14.0, 94.0, true},
- {OS_ALL, VENDOR_QUALCOMM, DRIVER_QUALCOMM, -1, BUG_BROKENCENTROID, 14.0, 46.0, true},
- {OS_ALL, VENDOR_QUALCOMM, DRIVER_QUALCOMM, -1, BUG_BROKENINFOLOG, -1.0, 46.0, true},
- {OS_ALL, VENDOR_QUALCOMM, DRIVER_QUALCOMM, -1, BUG_ANNIHILATEDUBOS, 41.0, 46.0, true},
- {OS_ALL, VENDOR_QUALCOMM, DRIVER_QUALCOMM, -1, BUG_BROKENSWAP, -1.0, 46.0, true},
{OS_ALL, VENDOR_QUALCOMM, DRIVER_QUALCOMM, -1, BUG_BROKENBUFFERSTREAM, -1.0, -1.0, true},
- {OS_ALL, VENDOR_QUALCOMM, DRIVER_QUALCOMM, -1, BUG_BROKENTEXTURESIZE, -1.0, 65.0, true},
- {OS_ALL, VENDOR_QUALCOMM, DRIVER_QUALCOMM, -1, BUG_BROKENATTRIBUTELESS, -1.0, 94.0, true},
{OS_ALL, VENDOR_QUALCOMM, DRIVER_QUALCOMM, -1, BUG_BROKENNEGATEDBOOLEAN,-1.0, -1.0, true},
- {OS_ALL, VENDOR_QUALCOMM, DRIVER_QUALCOMM, -1, BUG_BROKENIVECSHIFTS, -1.0, 46.0, true},
+ {OS_ALL, VENDOR_QUALCOMM, DRIVER_QUALCOMM, -1, BUG_BROKENGLES31, -1.0, -1.0, true},
{OS_ALL, VENDOR_ARM, DRIVER_ARM, -1, BUG_BROKENBUFFERSTREAM, -1.0, -1.0, true},
+ {OS_ALL, VENDOR_ARM, DRIVER_ARM, -1, BUG_BROKENVSYNC, -1.0, -1.0, true},
+ {OS_ALL, VENDOR_IMGTEC, DRIVER_IMGTEC, -1, BUG_BROKENBUFFERSTREAM, -1.0, -1.0, true},
{OS_ALL, VENDOR_MESA, DRIVER_NOUVEAU, -1, BUG_BROKENUBO, 900, 916, true},
{OS_ALL, VENDOR_MESA, DRIVER_R600, -1, BUG_BROKENUBO, 900, 913, true},
{OS_ALL, VENDOR_MESA, DRIVER_I965, -1, BUG_BROKENUBO, 900, 920, true},
+ {OS_ALL, VENDOR_MESA, DRIVER_ALL, -1, BUG_BROKENCOPYIMAGE, -1.0, 1064.0, true},
{OS_LINUX, VENDOR_ATI, DRIVER_ATI, -1, BUG_BROKENPINNEDMEMORY, -1.0, -1.0, true},
{OS_LINUX, VENDOR_NVIDIA, DRIVER_NVIDIA, -1, BUG_BROKENBUFFERSTORAGE, -1.0, 33138.0, true},
{OS_OSX, VENDOR_INTEL, DRIVER_INTEL, 3000, BUG_PRIMITIVERESTART, -1.0, -1.0, true},
@@ -105,7 +101,7 @@ namespace DriverDetails
( bug.m_versionstart <= m_version || bug.m_versionstart == -1 ) &&
( bug.m_versionend > m_version || bug.m_versionend == -1 )
)
- m_bugs.insert(std::make_pair(bug.m_bug, bug));
+ m_bugs.emplace(bug.m_bug, bug);
}
}
diff --git a/Source/Core/VideoCommon/DriverDetails.h b/Source/Core/VideoCommon/DriverDetails.h
index cdc5026dc2..1d49717499 100644
--- a/Source/Core/VideoCommon/DriverDetails.h
+++ b/Source/Core/VideoCommon/DriverDetails.h
@@ -58,32 +58,6 @@ namespace DriverDetails
// This'll ensure we know exactly what the issue is.
enum Bug
{
- // Bug: No Dynamic UBO array object access
- // Affected Devices: Qualcomm/Adreno
- // Started Version: 14
- // Ended Version: 95
- // Accessing UBO array members dynamically causes the Adreno shader compiler to crash
- // Errors out with "Internal Error"
- // With v53 video drivers, dynamic member access "works." It works to the extent that it doesn't crash.
- // With v95 drivers everything works as it should.
- BUG_NODYNUBOACCESS = 0,
- // Bug: Centroid is broken in shaders
- // Affected devices: Qualcomm/Adreno
- // Started Version: 14
- // Ended Version: 53
- // Centroid in/out, used in the shaders, is used for multisample buffers to get the texel correctly
- // When MSAA is disabled, it acts like a regular in/out
- // Tends to cause the driver to render full white or black
- BUG_BROKENCENTROID,
- // Bug: INFO_LOG_LENGTH broken
- // Affected devices: Qualcomm/Adreno
- // Started Version: ? (Noticed on v14)
- // Ended Version: 53
- // When compiling a shader, it is important that when it fails,
- // you first get the length of the information log prior to grabbing it.
- // This allows you to allocate an array to store all of the log
- // Adreno devices /always/ return 0 when querying GL_INFO_LOG_LENGTH
- // They also max out at 1024 bytes(1023 characters + null terminator) for the log
BUG_BROKENINFOLOG,
// Bug: UBO buffer offset broken
// Affected devices: all mesa drivers
@@ -104,22 +78,6 @@ namespace DriverDetails
// Please see issue #6105 on Google Code. Let's hope buffer storage solves this issues.
// TODO: Detect broken drivers.
BUG_BROKENPINNEDMEMORY,
- // Bug: Entirely broken UBOs
- // Affected devices: Qualcomm/Adreno
- // Started Version: ? (Noticed on v45)
- // Ended Version: 53
- // Uniform buffers are entirely broken on Qualcomm drivers with v45
- // Trying to use the uniform buffers causes a malloc to fail inside the driver
- // To be safe, blanket drivers from v41 - v45
- BUG_ANNIHILATEDUBOS,
- // Bug : Can't draw on screen text and clear correctly.
- // Affected devices: Qualcomm/Adreno
- // Started Version: ?
- // Ended Version: 53
- // Current code for drawing on screen text and clearing the framebuffer doesn't work on Adreno
- // Drawing on screen text causes the whole screen to swizzle in a terrible fashion
- // Clearing the framebuffer causes one to never see a frame.
- BUG_BROKENSWAP,
// Bug: glBufferSubData/glMapBufferRange stalls + OOM
// Affected devices: Adreno a3xx/Mali-t6xx
// Started Version: -1
@@ -128,12 +86,6 @@ namespace DriverDetails
// The driver stalls in each instance no matter what you do
// Apparently Mali and Adreno share code in this regard since it was wrote by the same person.
BUG_BROKENBUFFERSTREAM,
- // Bug: GLSL ES 3.0 textureSize causes abort
- // Affected devices: Adreno a3xx
- // Started Version: -1 (Noticed in v53)
- // Ended Version: 66
- // If a shader includes a textureSize function call then the shader compiler will call abort()
- BUG_BROKENTEXTURESIZE,
// Bug: ARB_buffer_storage doesn't work with ARRAY_BUFFER type streams
// Affected devices: GeForce 4xx+
// Started Version: -1
@@ -169,14 +121,6 @@ namespace DriverDetails
// It works for all the buffer types we use except GL_ELEMENT_ARRAY_BUFFER.
// Causes complete blackscreen issues.
BUG_INTELBROKENBUFFERSTORAGE,
- // Bug: Qualcomm has broken attributeless rendering
- // Affected devices: Adreno
- // Started Version: -1
- // Ended Version: v66 (07-09-2014 dev version), v95 shipping
- // Qualcomm has had attributeless rendering broken forever
- // This was fixed in a v66 development version, the first shipping driver version with the release was v95.
- // To be safe, make v95 the minimum version to work around this issue
- BUG_BROKENATTRIBUTELESS,
// Bug: Qualcomm has broken boolean negation
// Affected devices: Adreno
// Started Version: -1
@@ -202,38 +146,30 @@ namespace DriverDetails
// if (cond == false)
BUG_BROKENNEGATEDBOOLEAN,
- // Bug: Qualcomm has broken ivec to scalar and ivec to ivec bitshifts
+ // Bug: glCopyImageSubData doesn't work on i965
+ // Started Version: -1
+ // Ended Version: 10.6.4
+ // Mesa meta misses to disable the scissor test.
+ BUG_BROKENCOPYIMAGE,
+
+ // Bug: Qualcomm has broken OpenGL ES 3.1 support
// Affected devices: Adreno
// Started Version: -1
- // Ended Version: 46 (TODO: Test more devices, the real end is currently unknown)
- // Qualcomm has broken integer vector to integer bitshifts, and integer vector to integer vector bitshifts
- // A compilation error is generated when trying to compile the shaders.
- //
- // For example:
- // Broken on Qualcomm:
- // ivec4 ab = ivec4(1,1,1,1);
- // ab <<= 2;
- //
- // Working on Qualcomm:
- // ivec4 ab = ivec4(1,1,1,1);
- // ab.x <<= 2;
- // ab.y <<= 2;
- // ab.z <<= 2;
- // ab.w <<= 2;
- //
- // Broken on Qualcomm:
- // ivec4 ab = ivec4(1,1,1,1);
- // ivec4 cd = ivec4(1,2,3,4);
- // ab <<= cd;
- //
- // Working on Qualcomm:
- // ivec4 ab = ivec4(1,1,1,1);
- // ivec4 cd = ivec4(1,2,3,4);
- // ab.x <<= cd.x;
- // ab.y <<= cd.y;
- // ab.z <<= cd.z;
- // ab.w <<= cd.w;
- BUG_BROKENIVECSHIFTS,
+ // Ended Version: -1
+ // This isn't fully researched, but at the very least Qualcomm doesn't implement Geometry shader features fully.
+ // Until each bug is fully investigated, just disable GLES 3.1 entirely on these devices.
+ BUG_BROKENGLES31,
+
+ // Bug: ARM Mali managed to break disabling vsync
+ // Affected Devices: Mali
+ // Started Version: r5p0-rev2
+ // Ended Version: -1
+ // If we disable vsync with eglSwapInterval(dpy, 0) then the screen will stop showing new updates after a handful of swaps.
+ // This was noticed on a Samsung Galaxy S6 with its Android 5.1.1 update.
+ // The default Android 5.0 image didn't encounter this issue.
+ // We can't actually detect what the driver version is on Android, so until the driver version lands that displays the version in
+ // the GL_VERSION string, we will have to force vsync to be enabled at all times.
+ BUG_BROKENVSYNC,
};
// Initializes our internal vendor, device family, and driver version
diff --git a/Source/Core/VideoCommon/Fifo.cpp b/Source/Core/VideoCommon/Fifo.cpp
index a788d61ce7..2395d2b324 100644
--- a/Source/Core/VideoCommon/Fifo.cpp
+++ b/Source/Core/VideoCommon/Fifo.cpp
@@ -217,7 +217,7 @@ static void ReadDataFromFifo(u32 readPtr)
size_t existing_len = s_video_buffer_write_ptr - s_video_buffer_read_ptr;
if (len > (size_t)(FIFO_SIZE - existing_len))
{
- PanicAlert("FIFO out of bounds (existing %lu + new %lu > %lu)", (unsigned long) existing_len, (unsigned long) len, (unsigned long) FIFO_SIZE);
+ PanicAlert("FIFO out of bounds (existing %zu + new %zu > %lu)", existing_len, len, (unsigned long) FIFO_SIZE);
return;
}
memmove(s_video_buffer, s_video_buffer_read_ptr, existing_len);
@@ -254,7 +254,7 @@ static void ReadDataFromFifoOnCPU(u32 readPtr)
size_t existing_len = write_ptr - s_video_buffer_pp_read_ptr;
if (len > (size_t)(FIFO_SIZE - existing_len))
{
- PanicAlert("FIFO out of bounds (existing %lu + new %lu > %lu)", (unsigned long) existing_len, (unsigned long) len, (unsigned long) FIFO_SIZE);
+ PanicAlert("FIFO out of bounds (existing %zu + new %zu > %lu)", existing_len, len, (unsigned long) FIFO_SIZE);
return;
}
}
diff --git a/Source/Core/VideoCommon/FramebufferManagerBase.cpp b/Source/Core/VideoCommon/FramebufferManagerBase.cpp
index 10d1cfa414..7981dfce4d 100644
--- a/Source/Core/VideoCommon/FramebufferManagerBase.cpp
+++ b/Source/Core/VideoCommon/FramebufferManagerBase.cpp
@@ -2,7 +2,7 @@
// Licensed under GPLv2+
// Refer to the license.txt file included.
-
+#include <algorithm>
#include "VideoCommon/FramebufferManagerBase.h"
#include "VideoCommon/RenderBase.h"
#include "VideoCommon/VideoConfig.h"
@@ -115,17 +115,17 @@ const XFBSourceBase* const* FramebufferManagerBase::GetVirtualXFBSource(u32 xfbA
return &m_overlappingXFBArray[0];
}
-void FramebufferManagerBase::CopyToXFB(u32 xfbAddr, u32 fbWidth, u32 fbHeight, const EFBRectangle& sourceRc,float Gamma)
+void FramebufferManagerBase::CopyToXFB(u32 xfbAddr, u32 fbStride, u32 fbHeight, const EFBRectangle& sourceRc, float Gamma)
{
if (g_ActiveConfig.bUseRealXFB)
- g_framebuffer_manager->CopyToRealXFB(xfbAddr, fbWidth, fbHeight, sourceRc,Gamma);
+ g_framebuffer_manager->CopyToRealXFB(xfbAddr, fbStride, fbHeight, sourceRc, Gamma);
else
- CopyToVirtualXFB(xfbAddr, fbWidth, fbHeight, sourceRc,Gamma);
+ CopyToVirtualXFB(xfbAddr, fbStride, fbHeight, sourceRc, Gamma);
}
-void FramebufferManagerBase::CopyToVirtualXFB(u32 xfbAddr, u32 fbWidth, u32 fbHeight, const EFBRectangle& sourceRc,float Gamma)
+void FramebufferManagerBase::CopyToVirtualXFB(u32 xfbAddr, u32 fbStride, u32 fbHeight, const EFBRectangle& sourceRc, float Gamma)
{
- VirtualXFBListType::iterator vxfb = FindVirtualXFB(xfbAddr, fbWidth, fbHeight);
+ VirtualXFBListType::iterator vxfb = FindVirtualXFB(xfbAddr, sourceRc.GetWidth(), fbHeight);
if (m_virtualXFBList.end() == vxfb)
{
@@ -165,7 +165,7 @@ void FramebufferManagerBase::CopyToVirtualXFB(u32 xfbAddr, u32 fbWidth, u32 fbHe
}
vxfb->xfbSource->srcAddr = vxfb->xfbAddr = xfbAddr;
- vxfb->xfbSource->srcWidth = vxfb->xfbWidth = fbWidth;
+ vxfb->xfbSource->srcWidth = vxfb->xfbWidth = sourceRc.GetWidth();
vxfb->xfbSource->srcHeight = vxfb->xfbHeight = fbHeight;
vxfb->xfbSource->sourceRc = g_renderer->ConvertEFBRectangle(sourceRc);
@@ -182,17 +182,12 @@ FramebufferManagerBase::VirtualXFBListType::iterator FramebufferManagerBase::Fin
const u32 srcLower = xfbAddr;
const u32 srcUpper = xfbAddr + 2 * width * height;
- VirtualXFBListType::iterator it = m_virtualXFBList.begin();
- for (; it != m_virtualXFBList.end(); ++it)
- {
- const u32 dstLower = it->xfbAddr;
- const u32 dstUpper = it->xfbAddr + 2 * it->xfbWidth * it->xfbHeight;
-
- if (dstLower >= srcLower && dstUpper <= srcUpper)
- break;
- }
+ return std::find_if(m_virtualXFBList.begin(), m_virtualXFBList.end(), [srcLower, srcUpper](const VirtualXFB& xfb) {
+ const u32 dstLower = xfb.xfbAddr;
+ const u32 dstUpper = xfb.xfbAddr + 2 * xfb.xfbWidth * xfb.xfbHeight;
- return it;
+ return dstLower >= srcLower && dstUpper <= srcUpper;
+ });
}
void FramebufferManagerBase::ReplaceVirtualXFB()
diff --git a/Source/Core/VideoCommon/FramebufferManagerBase.h b/Source/Core/VideoCommon/FramebufferManagerBase.h
index bd6360877c..6127037268 100644
--- a/Source/Core/VideoCommon/FramebufferManagerBase.h
+++ b/Source/Core/VideoCommon/FramebufferManagerBase.h
@@ -45,7 +45,7 @@ public:
FramebufferManagerBase();
virtual ~FramebufferManagerBase();
- static void CopyToXFB(u32 xfbAddr, u32 fbWidth, u32 fbHeight, const EFBRectangle& sourceRc,float Gamma);
+ static void CopyToXFB(u32 xfbAddr, u32 fbStride, u32 fbHeight, const EFBRectangle& sourceRc,float Gamma);
static const XFBSourceBase* const* GetXFBSource(u32 xfbAddr, u32 fbWidth, u32 fbHeight, u32* xfbCount);
static void SetLastXfbWidth(unsigned int width) { s_last_xfb_width = width; }
@@ -87,7 +87,7 @@ private:
static void ReplaceVirtualXFB();
// TODO: merge these virtual funcs, they are nearly all the same
- virtual void CopyToRealXFB(u32 xfbAddr, u32 fbWidth, u32 fbHeight, const EFBRectangle& sourceRc,float Gamma = 1.0f) = 0;
+ virtual void CopyToRealXFB(u32 xfbAddr, u32 fbStride, u32 fbHeight, const EFBRectangle& sourceRc,float Gamma = 1.0f) = 0;
static void CopyToVirtualXFB(u32 xfbAddr, u32 fbWidth, u32 fbHeight, const EFBRectangle& sourceRc,float Gamma = 1.0f);
static const XFBSourceBase* const* GetRealXFBSource(u32 xfbAddr, u32 fbWidth, u32 fbHeight, u32* xfbCount);
diff --git a/Source/Core/VideoCommon/GeometryShaderGen.cpp b/Source/Core/VideoCommon/GeometryShaderGen.cpp
index 5ec98b3813..f36aa94a9a 100644
--- a/Source/Core/VideoCommon/GeometryShaderGen.cpp
+++ b/Source/Core/VideoCommon/GeometryShaderGen.cpp
@@ -94,11 +94,11 @@ static inline void GenerateGeometryShader(T& out, u32 primitive_type, API_TYPE A
out.Write("#define InstanceID gl_InvocationID\n");
out.Write("in VertexData {\n");
- GenerateVSOutputMembers<T>(out, ApiType, g_ActiveConfig.backend_info.bSupportsBindingLayout ? "centroid" : "centroid in");
+ GenerateVSOutputMembers<T>(out, ApiType, GetInterpolationQualifier(ApiType, true, true));
out.Write("} vs[%d];\n", vertex_in);
out.Write("out VertexData {\n");
- GenerateVSOutputMembers<T>(out, ApiType, g_ActiveConfig.backend_info.bSupportsBindingLayout ? "centroid" : "centroid out");
+ GenerateVSOutputMembers<T>(out, ApiType, GetInterpolationQualifier(ApiType, false, true));
if (g_ActiveConfig.iStereoMode > 0)
out.Write("\tflat int layer;\n");
diff --git a/Source/Core/VideoCommon/HiresTextures.cpp b/Source/Core/VideoCommon/HiresTextures.cpp
index efe3ab58bc..e6f687a551 100644
--- a/Source/Core/VideoCommon/HiresTextures.cpp
+++ b/Source/Core/VideoCommon/HiresTextures.cpp
@@ -140,7 +140,10 @@ void HiresTexture::Prefetch()
Common::SetCurrentThreadName("Prefetcher");
size_t size_sum = 0;
- size_t max_mem = MemPhysical() / 2;
+ size_t sys_mem = MemPhysical();
+ size_t recommended_min_mem = 2 * size_t(1024 * 1024 * 1024);
+ // keep 2GB memory for system stability if system RAM is 4GB+ - use half of memory in other cases
+ size_t max_mem = (sys_mem / 2 < recommended_min_mem) ? (sys_mem / 2) : (sys_mem - recommended_min_mem);
u32 starttime = Common::Timer::GetTimeMs();
for (const auto& entry : s_textureMap)
{
diff --git a/Source/Core/VideoCommon/ImageWrite.cpp b/Source/Core/VideoCommon/ImageWrite.cpp
index 38ca509c3e..f50c5d02b4 100644
--- a/Source/Core/VideoCommon/ImageWrite.cpp
+++ b/Source/Core/VideoCommon/ImageWrite.cpp
@@ -8,6 +8,8 @@
#include "png.h"
#include "Common/FileUtil.h"
+#include "Common/MsgHandler.h"
+#include "Common/Logging/Log.h"
#include "VideoCommon/ImageWrite.h"
bool SaveData(const std::string& filename, const char* data)
diff --git a/Source/Core/VideoCommon/IndexGenerator.cpp b/Source/Core/VideoCommon/IndexGenerator.cpp
index 19a6642f66..c1941c09b8 100644
--- a/Source/Core/VideoCommon/IndexGenerator.cpp
+++ b/Source/Core/VideoCommon/IndexGenerator.cpp
@@ -4,6 +4,7 @@
#include <cstddef>
+#include "Common/Common.h"
#include "Common/CommonTypes.h"
#include "VideoCommon/IndexGenerator.h"
#include "VideoCommon/OpcodeDecoding.h"
diff --git a/Source/Core/VideoCommon/LightingShaderGen.h b/Source/Core/VideoCommon/LightingShaderGen.h
index 83ecfb4174..ae388ac5ed 100644
--- a/Source/Core/VideoCommon/LightingShaderGen.h
+++ b/Source/Core/VideoCommon/LightingShaderGen.h
@@ -253,10 +253,7 @@ static void GenerateLightingShader(T& object, LightingUidData& uid_data, int com
}
}
object.Write("lacc = clamp(lacc, 0, 255);\n");
- if (DriverDetails::HasBug(DriverDetails::BUG_BROKENIVECSHIFTS))
- object.Write("%s%d = float4(irshift((mat * (lacc + irshift(lacc, 7))), 8)) / 255.0;\n", dest, j);
- else
- object.Write("%s%d = float4((mat * (lacc + (lacc >> 7))) >> 8) / 255.0;\n", dest, j);
+ object.Write("%s%d = float4((mat * (lacc + (lacc >> 7))) >> 8) / 255.0;\n", dest, j);
object.Write("}\n");
}
}
diff --git a/Source/Core/VideoCommon/LookUpTables.h b/Source/Core/VideoCommon/LookUpTables.h
index d821597c8f..4bda199be5 100644
--- a/Source/Core/VideoCommon/LookUpTables.h
+++ b/Source/Core/VideoCommon/LookUpTables.h
@@ -6,25 +6,25 @@
#include "Common/CommonTypes.h"
-inline u8 Convert3To8(u8 v)
+constexpr u8 Convert3To8(u8 v)
{
// Swizzle bits: 00000123 -> 12312312
return (v << 5) | (v << 2) | (v >> 1);
}
-inline u8 Convert4To8(u8 v)
+constexpr u8 Convert4To8(u8 v)
{
// Swizzle bits: 00001234 -> 12341234
return (v << 4) | v;
}
-inline u8 Convert5To8(u8 v)
+constexpr u8 Convert5To8(u8 v)
{
// Swizzle bits: 00012345 -> 12345123
return (v << 3) | (v >> 2);
}
-inline u8 Convert6To8(u8 v)
+constexpr u8 Convert6To8(u8 v)
{
// Swizzle bits: 00123456 -> 12345612
return (v << 2) | (v >> 4);
diff --git a/Source/Core/VideoCommon/NativeVertexFormat.h b/Source/Core/VideoCommon/NativeVertexFormat.h
index c64d94fa77..df86b90946 100644
--- a/Source/Core/VideoCommon/NativeVertexFormat.h
+++ b/Source/Core/VideoCommon/NativeVertexFormat.h
@@ -4,10 +4,12 @@
#pragma once
+#include <cstring>
#include <functional> // for hash
-#include "Common/Common.h"
+#include "Common/CommonTypes.h"
#include "Common/Hash.h"
+#include "Common/NonCopyable.h"
// m_components
enum
diff --git a/Source/Core/VideoCommon/OnScreenDisplay.cpp b/Source/Core/VideoCommon/OnScreenDisplay.cpp
index ddd80065e1..dfd597f0ed 100644
--- a/Source/Core/VideoCommon/OnScreenDisplay.cpp
+++ b/Source/Core/VideoCommon/OnScreenDisplay.cpp
@@ -71,7 +71,7 @@ void ClearMessages()
// On-Screen Display Callbacks
void AddCallback(CallbackType type, Callback cb)
{
- s_callbacks.insert(std::pair<CallbackType, Callback>(type, cb));
+ s_callbacks.emplace(type, cb);
}
void DoCallbacks(CallbackType type)
diff --git a/Source/Core/VideoCommon/OpcodeDecoding.h b/Source/Core/VideoCommon/OpcodeDecoding.h
index ec1b2652e4..a79bf54be2 100644
--- a/Source/Core/VideoCommon/OpcodeDecoding.h
+++ b/Source/Core/VideoCommon/OpcodeDecoding.h
@@ -38,8 +38,6 @@
#define GX_DRAW_LINE_STRIP 0x6 // 0xB0
#define GX_DRAW_POINTS 0x7 // 0xB8
-extern bool g_bRecordFifoData;
-
void OpcodeDecoder_Init();
void OpcodeDecoder_Shutdown();
diff --git a/Source/Core/VideoCommon/PixelShaderGen.cpp b/Source/Core/VideoCommon/PixelShaderGen.cpp
index ad251203f9..2a65a32836 100644
--- a/Source/Core/VideoCommon/PixelShaderGen.cpp
+++ b/Source/Core/VideoCommon/PixelShaderGen.cpp
@@ -6,6 +6,7 @@
#include <cmath>
#include <cstdio>
+#include "Common/Common.h"
#include "VideoCommon/BoundingBox.h"
#include "VideoCommon/BPMemory.h"
#include "VideoCommon/ConstantManager.h"
@@ -17,6 +18,24 @@
#include "VideoCommon/VideoConfig.h"
#include "VideoCommon/XFMemory.h" // for texture projection mode
+// TODO: Get rid of these
+enum : u32
+{
+ C_COLORMATRIX = 0, // 0
+ C_COLORS = 0, // 0
+ C_KCOLORS = C_COLORS + 4, // 4
+ C_ALPHA = C_KCOLORS + 4, // 8
+ C_TEXDIMS = C_ALPHA + 1, // 9
+ C_ZBIAS = C_TEXDIMS + 8, // 17
+ C_INDTEXSCALE = C_ZBIAS + 2, // 19
+ C_INDTEXMTX = C_INDTEXSCALE + 2, // 21
+ C_FOGCOLOR = C_INDTEXMTX + 6, // 27
+ C_FOGI = C_FOGCOLOR + 1, // 28
+ C_FOGF = C_FOGI + 1, // 29
+ C_ZSLOPE = C_FOGF + 2, // 31
+ C_EFBSCALE = C_ZSLOPE + 1, // 32
+ C_PENVCONST_END = C_EFBSCALE + 1
+};
static const char *tevKSelTableC[] =
{
@@ -137,7 +156,7 @@ static const char *tevRasTable[] =
static const char *tevCOutputTable[] = { "prev.rgb", "c0.rgb", "c1.rgb", "c2.rgb" };
static const char *tevAOutputTable[] = { "prev.a", "c0.a", "c1.a", "c2.a" };
-static char text[16384];
+static char text[32768];
template<class T> static inline void WriteStage(T& out, pixel_shader_uid_data* uid_data, int n, API_TYPE ApiType, const char swapModeTable[4][5]);
template<class T> static inline void WriteTevRegular(T& out, const char* components, int bias, int op, int clamp, int shift);
@@ -196,30 +215,6 @@ static inline void GeneratePixelShader(T& out, DSTALPHA_MODE dstAlphaMode, API_T
"int3 itrunc(float3 x) { return int3(trunc(x)); }\n"
"int4 itrunc(float4 x) { return int4(trunc(x)); }\n\n");
- if (DriverDetails::HasBug(DriverDetails::BUG_BROKENIVECSHIFTS))
- {
- // Add functions to do shifts on scalars and ivecs.
- // These functions all have the same name to enable them to be used no matter what code is generated.
- // For example: tev color op code uses .rgb as a swizzle, but alpha code only uses .a.
- out.Write("int ilshift(int a, int b) { return a << b; }\n"
- "int irshift(int a, int b) { return a >> b; }\n"
-
- "int2 ilshift(int2 a, int2 b) { return int2(a.x << b.x, a.y << b.y); }\n"
- "int2 ilshift(int2 a, int b) { return int2(a.x << b, a.y << b); }\n"
- "int2 irshift(int2 a, int2 b) { return int2(a.x >> b.x, a.y >> b.y); }\n"
- "int2 irshift(int2 a, int b) { return int2(a.x >> b, a.y >> b); }\n"
-
- "int3 ilshift(int3 a, int3 b) { return int3(a.x << b.x, a.y << b.y, a.z << b.z); }\n"
- "int3 ilshift(int3 a, int b) { return int3(a.x << b, a.y << b, a.z << b); }\n"
- "int3 irshift(int3 a, int3 b) { return int3(a.x >> b.x, a.y >> b.y, a.z >> b.z); }\n"
- "int3 irshift(int3 a, int b) { return int3(a.x >> b, a.y >> b, a.z >> b); }\n"
-
- "int4 ilshift(int4 a, int4 b) { return int4(a.x << b.x, a.y << b.y, a.z << b.z, a.w << b.w); }\n"
- "int4 ilshift(int4 a, int b) { return int4(a.x << b, a.y << b, a.z << b, a.w << b); }\n"
- "int4 irshift(int4 a, int4 b) { return int4(a.x >> b.x, a.y >> b.y, a.z >> b.z, a.w >> b.w); }\n"
- "int4 irshift(int4 a, int b) { return int4(a.x >> b, a.y >> b, a.z >> b, a.w >> b); }\n\n");
- }
-
if (ApiType == API_OPENGL)
{
// Declare samplers
@@ -338,6 +333,8 @@ static inline void GeneratePixelShader(T& out, DSTALPHA_MODE dstAlphaMode, API_T
warn_once = false;
}
+ uid_data->msaa = g_ActiveConfig.iMultisampleMode > 0;
+ uid_data->ssaa = g_ActiveConfig.iMultisampleMode > 0 && g_ActiveConfig.bSSAA;
if (ApiType == API_OPENGL)
{
out.Write("out vec4 ocol0;\n");
@@ -347,19 +344,11 @@ static inline void GeneratePixelShader(T& out, DSTALPHA_MODE dstAlphaMode, API_T
if (per_pixel_depth)
out.Write("#define depth gl_FragDepth\n");
- // We use the flag "centroid" to fix some MSAA rendering bugs. With MSAA, the
- // pixel shader will be executed for each pixel which has at least one passed sample.
- // So there may be rendered pixels where the center of the pixel isn't in the primitive.
- // As the pixel shader usually renders at the center of the pixel, this position may be
- // outside the primitive. This will lead to sampling outside the texture, sign changes, ...
- // As a workaround, we interpolate at the centroid of the coveraged pixel, which
- // is always inside the primitive.
- // Without MSAA, this flag is defined to have no effect.
uid_data->stereo = g_ActiveConfig.iStereoMode > 0;
if (g_ActiveConfig.backend_info.bSupportsGeometryShaders)
{
out.Write("in VertexData {\n");
- GenerateVSOutputMembers<T>(out, ApiType, g_ActiveConfig.backend_info.bSupportsBindingLayout ? "centroid" : "centroid in");
+ GenerateVSOutputMembers<T>(out, ApiType, GetInterpolationQualifier(ApiType, true, true));
if (g_ActiveConfig.iStereoMode > 0)
out.Write("\tflat int layer;\n");
@@ -368,19 +357,19 @@ static inline void GeneratePixelShader(T& out, DSTALPHA_MODE dstAlphaMode, API_T
}
else
{
- out.Write("centroid in float4 colors_0;\n");
- out.Write("centroid in float4 colors_1;\n");
+ out.Write("%s in float4 colors_0;\n", GetInterpolationQualifier(ApiType));
+ out.Write("%s in float4 colors_1;\n", GetInterpolationQualifier(ApiType));
// compute window position if needed because binding semantic WPOS is not widely supported
// Let's set up attributes
for (unsigned int i = 0; i < numTexgen; ++i)
{
- out.Write("centroid in float3 uv%d;\n", i);
+ out.Write("%s in float3 uv%d;\n", GetInterpolationQualifier(ApiType), i);
}
- out.Write("centroid in float4 clipPos;\n");
+ out.Write("%s in float4 clipPos;\n", GetInterpolationQualifier(ApiType));
if (g_ActiveConfig.bEnablePixelLighting)
{
- out.Write("centroid in float3 Normal;\n");
- out.Write("centroid in float3 WorldPos;\n");
+ out.Write("%s in float3 Normal;\n", GetInterpolationQualifier(ApiType));
+ out.Write("%s in float3 WorldPos;\n", GetInterpolationQualifier(ApiType));
}
}
@@ -401,17 +390,17 @@ static inline void GeneratePixelShader(T& out, DSTALPHA_MODE dstAlphaMode, API_T
dstAlphaMode == DSTALPHA_DUAL_SOURCE_BLEND ? "\n out float4 ocol1 : SV_Target1," : "",
per_pixel_depth ? "\n out float depth : SV_Depth," : "");
- out.Write(" in centroid float4 colors_0 : COLOR0,\n");
- out.Write(" in centroid float4 colors_1 : COLOR1\n");
+ out.Write(" in %s float4 colors_0 : COLOR0,\n", GetInterpolationQualifier(ApiType));
+ out.Write(" in %s float4 colors_1 : COLOR1\n", GetInterpolationQualifier(ApiType));
// compute window position if needed because binding semantic WPOS is not widely supported
for (unsigned int i = 0; i < numTexgen; ++i)
- out.Write(",\n in centroid float3 uv%d : TEXCOORD%d", i, i);
- out.Write(",\n in centroid float4 clipPos : TEXCOORD%d", numTexgen);
+ out.Write(",\n in %s float3 uv%d : TEXCOORD%d", GetInterpolationQualifier(ApiType), i, i);
+ out.Write(",\n in %s float4 clipPos : TEXCOORD%d", GetInterpolationQualifier(ApiType), numTexgen);
if (g_ActiveConfig.bEnablePixelLighting)
{
- out.Write(",\n in centroid float3 Normal : TEXCOORD%d", numTexgen + 1);
- out.Write(",\n in centroid float3 WorldPos : TEXCOORD%d", numTexgen + 2);
+ out.Write(",\n in %s float3 Normal : TEXCOORD%d", GetInterpolationQualifier(ApiType), numTexgen + 1);
+ out.Write(",\n in %s float3 WorldPos : TEXCOORD%d", GetInterpolationQualifier(ApiType), numTexgen + 2);
}
uid_data->stereo = g_ActiveConfig.iStereoMode > 0;
if (g_ActiveConfig.iStereoMode > 0)
@@ -499,11 +488,7 @@ static inline void GeneratePixelShader(T& out, DSTALPHA_MODE dstAlphaMode, API_T
if (texcoord < numTexgen)
{
out.SetConstantsUsed(C_INDTEXSCALE+i/2,C_INDTEXSCALE+i/2);
-
- if (DriverDetails::HasBug(DriverDetails::BUG_BROKENIVECSHIFTS))
- out.Write("\ttempcoord = irshift(fixpoint_uv%d, " I_INDTEXSCALE"[%d].%s);\n", texcoord, i / 2, (i & 1) ? "zw" : "xy");
- else
- out.Write("\ttempcoord = fixpoint_uv%d >> " I_INDTEXSCALE"[%d].%s;\n", texcoord, i / 2, (i & 1) ? "zw" : "xy");
+ out.Write("\ttempcoord = fixpoint_uv%d >> " I_INDTEXSCALE"[%d].%s;\n", texcoord, i / 2, (i & 1) ? "zw" : "xy");
}
else
out.Write("\ttempcoord = int2(0, 0);\n");
@@ -729,22 +714,11 @@ static inline void WriteStage(T& out, pixel_shader_uid_data* uid_data, int n, AP
int mtxidx = 2*(bpmem.tevind[n].mid-1);
out.SetConstantsUsed(C_INDTEXMTX+mtxidx, C_INDTEXMTX+mtxidx);
- if (DriverDetails::HasBug(DriverDetails::BUG_BROKENIVECSHIFTS))
- {
- out.Write("\tint2 indtevtrans%d = irshift(int2(idot(" I_INDTEXMTX"[%d].xyz, iindtevcrd%d), idot(" I_INDTEXMTX"[%d].xyz, iindtevcrd%d)), 3);\n", n, mtxidx, n, mtxidx+1, n);
-
- // TODO: should use a shader uid branch for this for better performance
- out.Write("\tif (" I_INDTEXMTX"[%d].w >= 0) indtevtrans%d = irshift(indtevtrans%d, " I_INDTEXMTX"[%d].w);\n", mtxidx, n, n, mtxidx);
- out.Write("\telse indtevtrans%d = ilshift(indtevtrans%d, -" I_INDTEXMTX"[%d].w);\n", n, n, mtxidx);
- }
- else
- {
- out.Write("\tint2 indtevtrans%d = int2(idot(" I_INDTEXMTX"[%d].xyz, iindtevcrd%d), idot(" I_INDTEXMTX"[%d].xyz, iindtevcrd%d)) >> 3;\n", n, mtxidx, n, mtxidx+1, n);
-
- // TODO: should use a shader uid branch for this for better performance
- out.Write("\tif (" I_INDTEXMTX"[%d].w >= 0) indtevtrans%d = indtevtrans%d >> " I_INDTEXMTX"[%d].w;\n", mtxidx, n, n, mtxidx);
- out.Write("\telse indtevtrans%d = indtevtrans%d << (-" I_INDTEXMTX"[%d].w);\n", n, n, mtxidx);
- }
+ out.Write("\tint2 indtevtrans%d = int2(idot(" I_INDTEXMTX"[%d].xyz, iindtevcrd%d), idot(" I_INDTEXMTX"[%d].xyz, iindtevcrd%d)) >> 3;\n", n, mtxidx, n, mtxidx+1, n);
+
+ // TODO: should use a shader uid branch for this for better performance
+ out.Write("\tif (" I_INDTEXMTX"[%d].w >= 0) indtevtrans%d = indtevtrans%d >> " I_INDTEXMTX"[%d].w;\n", mtxidx, n, n, mtxidx);
+ out.Write("\telse indtevtrans%d = indtevtrans%d << (-" I_INDTEXMTX"[%d].w);\n", n, n, mtxidx);
}
else if (bpmem.tevind[n].mid <= 7 && bHasTexCoord)
{ // s matrix
@@ -752,20 +726,10 @@ static inline void WriteStage(T& out, pixel_shader_uid_data* uid_data, int n, AP
int mtxidx = 2*(bpmem.tevind[n].mid-5);
out.SetConstantsUsed(C_INDTEXMTX+mtxidx, C_INDTEXMTX+mtxidx);
- if (DriverDetails::HasBug(DriverDetails::BUG_BROKENIVECSHIFTS))
- {
- out.Write("\tint2 indtevtrans%d = irshift(int2(fixpoint_uv%d * iindtevcrd%d.xx), 8);\n", n, texcoord, n);
-
- out.Write("\tif (" I_INDTEXMTX"[%d].w >= 0) indtevtrans%d = irshift(indtevtrans%d, " I_INDTEXMTX"[%d].w);\n", mtxidx, n, n, mtxidx);
- out.Write("\telse indtevtrans%d = ilshift(indtevtrans%d, -" I_INDTEXMTX"[%d].w);\n", n, n, mtxidx);
- }
- else
- {
- out.Write("\tint2 indtevtrans%d = int2(fixpoint_uv%d * iindtevcrd%d.xx) >> 8;\n", n, texcoord, n);
+ out.Write("\tint2 indtevtrans%d = int2(fixpoint_uv%d * iindtevcrd%d.xx) >> 8;\n", n, texcoord, n);
- out.Write("\tif (" I_INDTEXMTX"[%d].w >= 0) indtevtrans%d = indtevtrans%d >> " I_INDTEXMTX"[%d].w;\n", mtxidx, n, n, mtxidx);
- out.Write("\telse indtevtrans%d = indtevtrans%d << (-" I_INDTEXMTX"[%d].w);\n", n, n, mtxidx);
- }
+ out.Write("\tif (" I_INDTEXMTX"[%d].w >= 0) indtevtrans%d = indtevtrans%d >> " I_INDTEXMTX"[%d].w;\n", mtxidx, n, n, mtxidx);
+ out.Write("\telse indtevtrans%d = indtevtrans%d << (-" I_INDTEXMTX"[%d].w);\n", n, n, mtxidx);
}
else if (bpmem.tevind[n].mid <= 11 && bHasTexCoord)
{ // t matrix
@@ -773,20 +737,10 @@ static inline void WriteStage(T& out, pixel_shader_uid_data* uid_data, int n, AP
int mtxidx = 2*(bpmem.tevind[n].mid-9);
out.SetConstantsUsed(C_INDTEXMTX+mtxidx, C_INDTEXMTX+mtxidx);
- if (DriverDetails::HasBug(DriverDetails::BUG_BROKENIVECSHIFTS))
- {
- out.Write("\tint2 indtevtrans%d = irshift(int2(fixpoint_uv%d * iindtevcrd%d.yy), 8);\n", n, texcoord, n);
+ out.Write("\tint2 indtevtrans%d = int2(fixpoint_uv%d * iindtevcrd%d.yy) >> 8;\n", n, texcoord, n);
- out.Write("\tif (" I_INDTEXMTX"[%d].w >= 0) indtevtrans%d = irshift(indtevtrans%d, " I_INDTEXMTX"[%d].w);\n", mtxidx, n, n, mtxidx);
- out.Write("\telse indtevtrans%d = ilshift(indtevtrans%d, -" I_INDTEXMTX"[%d].w);\n", n, n, mtxidx);
- }
- else
- {
- out.Write("\tint2 indtevtrans%d = int2(fixpoint_uv%d * iindtevcrd%d.yy) >> 8;\n", n, texcoord, n);
-
- out.Write("\tif (" I_INDTEXMTX"[%d].w >= 0) indtevtrans%d = indtevtrans%d >> " I_INDTEXMTX"[%d].w;\n", mtxidx, n, n, mtxidx);
- out.Write("\telse indtevtrans%d = indtevtrans%d << (-" I_INDTEXMTX"[%d].w);\n", n, n, mtxidx);
- }
+ out.Write("\tif (" I_INDTEXMTX"[%d].w >= 0) indtevtrans%d = indtevtrans%d >> " I_INDTEXMTX"[%d].w;\n", mtxidx, n, n, mtxidx);
+ out.Write("\telse indtevtrans%d = indtevtrans%d << (-" I_INDTEXMTX"[%d].w);\n", n, n, mtxidx);
}
else
{
@@ -825,10 +779,7 @@ static inline void WriteStage(T& out, pixel_shader_uid_data* uid_data, int n, AP
out.Write("\ttevcoord.xy = wrappedcoord + indtevtrans%d;\n", n);
// Emulate s24 overflows
- if (DriverDetails::HasBug(DriverDetails::BUG_BROKENIVECSHIFTS))
- out.Write("\ttevcoord.xy = irshift(ilshift(tevcoord.xy, 8), 8);\n");
- else
- out.Write("\ttevcoord.xy = (tevcoord.xy << 8) >> 8;\n");
+ out.Write("\ttevcoord.xy = (tevcoord.xy << 8) >> 8;\n");
}
TevStageCombiner::ColorCombiner &cc = bpmem.combiners[n].colorC;
@@ -923,14 +874,14 @@ static inline void WriteStage(T& out, pixel_shader_uid_data* uid_data, int n, AP
out.SetConstantsUsed(C_COLORS+ac.dest, C_COLORS+ac.dest);
- out.Write("\ttevin_a = int4(%s, %s)&255;\n", tevCInputTable[cc.a], tevAInputTable[ac.a]);
- out.Write("\ttevin_b = int4(%s, %s)&255;\n", tevCInputTable[cc.b], tevAInputTable[ac.b]);
- out.Write("\ttevin_c = int4(%s, %s)&255;\n", tevCInputTable[cc.c], tevAInputTable[ac.c]);
+ out.Write("\ttevin_a = int4(%s, %s)&int4(255, 255, 255, 255);\n", tevCInputTable[cc.a], tevAInputTable[ac.a]);
+ out.Write("\ttevin_b = int4(%s, %s)&int4(255, 255, 255, 255);\n", tevCInputTable[cc.b], tevAInputTable[ac.b]);
+ out.Write("\ttevin_c = int4(%s, %s)&int4(255, 255, 255, 255);\n", tevCInputTable[cc.c], tevAInputTable[ac.c]);
out.Write("\ttevin_d = int4(%s, %s);\n", tevCInputTable[cc.d], tevAInputTable[ac.d]);
out.Write("\t// color combine\n");
out.Write("\t%s = clamp(", tevCOutputTable[cc.dest]);
- if (cc.bias != TevBias_COMPARE)
+ if (cc.bias != TEVBIAS_COMPARE)
{
WriteTevRegular(out, "rgb", cc.bias, cc.op, cc.clamp, cc.shift);
}
@@ -960,7 +911,7 @@ static inline void WriteStage(T& out, pixel_shader_uid_data* uid_data, int n, AP
out.Write("\t// alpha combine\n");
out.Write("\t%s = clamp(", tevAOutputTable[ac.dest]);
- if (ac.bias != TevBias_COMPARE)
+ if (ac.bias != TEVBIAS_COMPARE)
{
WriteTevRegular(out, "a", ac.bias, ac.op, ac.clamp, ac.shift);
}
@@ -1035,37 +986,12 @@ static inline void WriteTevRegular(T& out, const char* components, int bias, int
// - c is scaled from 0..255 to 0..256, which allows dividing the result by 256 instead of 255
// - if scale is bigger than one, it is moved inside the lerp calculation for increased accuracy
// - a rounding bias is added before dividing by 256
-
- if (DriverDetails::HasBug(DriverDetails::BUG_BROKENIVECSHIFTS))
- {
- // Haxx - cleaner code by not having irshift and ilshift in the emitted code by omitting them if not used.
- const char* leftShift = tevScaleTableLeft[shift];
- const char* rightShift = tevScaleTableRight[shift];
-
- if (rightShift[0])
- out.Write("irshift(((tevin_d.%s%s)%s)", components, tevBiasTable[bias], tevScaleTableLeft[shift]);
- else
- out.Write("((tevin_d.%s%s)%s)", components, tevBiasTable[bias], tevScaleTableLeft[shift]);
- out.Write(" %s ", tevOpTable[op]);
- if (leftShift[0])
- out.Write("irshift((ilshift((ilshift(tevin_a.%s, 8) + (tevin_b.%s-tevin_a.%s)*(tevin_c.%s+irshift(tevin_c.%s, 7))), %s)%s), 8)",
- components, components, components, components, components,
- leftShift+4, tevLerpBias[2*op+(shift!=3)]);
- else
- out.Write("irshift(((ilshift(tevin_a.%s, 8) + (tevin_b.%s-tevin_a.%s)*(tevin_c.%s+irshift(tevin_c.%s, 7)))%s), 8)",
- components, components, components, components, components, tevLerpBias[2*op+(shift!=3)]);
- if (rightShift[0])
- out.Write(", %s)", rightShift+4);
- }
- else
- {
- out.Write("(((tevin_d.%s%s)%s)", components, tevBiasTable[bias], tevScaleTableLeft[shift]);
- out.Write(" %s ", tevOpTable[op]);
- out.Write("(((((tevin_a.%s<<8) + (tevin_b.%s-tevin_a.%s)*(tevin_c.%s+(tevin_c.%s>>7)))%s)%s)>>8)",
- components, components, components, components, components,
- tevScaleTableLeft[shift], tevLerpBias[2*op+(shift!=3)]);
- out.Write(")%s", tevScaleTableRight[shift]);
- }
+ out.Write("(((tevin_d.%s%s)%s)", components, tevBiasTable[bias], tevScaleTableLeft[shift]);
+ out.Write(" %s ", tevOpTable[op]);
+ out.Write("(((((tevin_a.%s<<8) + (tevin_b.%s-tevin_a.%s)*(tevin_c.%s+(tevin_c.%s>>7)))%s)%s)>>8)",
+ components, components, components, components, components,
+ tevScaleTableLeft[shift], tevLerpBias[2*op+(shift!=3)]);
+ out.Write(")%s", tevScaleTableRight[shift]);
}
template<class T>
@@ -1081,14 +1007,14 @@ static inline void SampleTexture(T& out, const char *texcoords, const char *texs
static const char *tevAlphaFuncsTable[] =
{
- "(false)", // NEVER
- "(prev.a < %s)", // LESS
- "(prev.a == %s)", // EQUAL
- "(prev.a <= %s)", // LEQUAL
- "(prev.a > %s)", // GREATER
- "(prev.a != %s)", // NEQUAL
- "(prev.a >= %s)", // GEQUAL
- "(true)" // ALWAYS
+ "(false)", // NEVER
+ "(prev.a < %s)", // LESS
+ "(prev.a == %s)", // EQUAL
+ "(prev.a <= %s)", // LEQUAL
+ "(prev.a > %s)", // GREATER
+ "(prev.a != %s)", // NEQUAL
+ "(prev.a >= %s)", // GEQUAL
+ "(true)" // ALWAYS
};
static const char *tevAlphaFunclogicTable[] =
@@ -1228,10 +1154,7 @@ static inline void WriteFog(T& out, pixel_shader_uid_data* uid_data)
}
out.Write("\tint ifog = iround(fog * 256.0);\n");
- if (DriverDetails::HasBug(DriverDetails::BUG_BROKENIVECSHIFTS))
- out.Write("\tprev.rgb = irshift((prev.rgb * (256 - ifog) + " I_FOGCOLOR".rgb * ifog), 8);\n");
- else
- out.Write("\tprev.rgb = (prev.rgb * (256 - ifog) + " I_FOGCOLOR".rgb * ifog) >> 8;\n");
+ out.Write("\tprev.rgb = (prev.rgb * (256 - ifog) + " I_FOGCOLOR".rgb * ifog) >> 8;\n");
}
void GetPixelShaderUid(PixelShaderUid& object, DSTALPHA_MODE dstAlphaMode, API_TYPE ApiType, u32 components)
diff --git a/Source/Core/VideoCommon/PixelShaderGen.h b/Source/Core/VideoCommon/PixelShaderGen.h
index 5a871fce6d..518207f7bc 100644
--- a/Source/Core/VideoCommon/PixelShaderGen.h
+++ b/Source/Core/VideoCommon/PixelShaderGen.h
@@ -9,23 +9,6 @@
#include "VideoCommon/ShaderGenCommon.h"
#include "VideoCommon/VideoCommon.h"
-// TODO: get rid of them as they aren't used
-#define C_COLORMATRIX 0 // 0
-#define C_COLORS 0 // 0
-#define C_KCOLORS (C_COLORS + 4) // 4
-#define C_ALPHA (C_KCOLORS + 4) // 8
-#define C_TEXDIMS (C_ALPHA + 1) // 9
-#define C_ZBIAS (C_TEXDIMS + 8) //17
-#define C_INDTEXSCALE (C_ZBIAS + 2) //19
-#define C_INDTEXMTX (C_INDTEXSCALE + 2) //21
-#define C_FOGCOLOR (C_INDTEXMTX + 6) //27
-#define C_FOGI (C_FOGCOLOR + 1) //28
-#define C_FOGF (C_FOGI + 1) //29
-#define C_ZSLOPE (C_FOGF + 2) //31
-#define C_EFBSCALE (C_ZSLOPE + 1) //32
-
-#define C_PENVCONST_END (C_EFBSCALE + 1)
-
// Different ways to achieve rendering with destination alpha
enum DSTALPHA_MODE
{
@@ -65,9 +48,11 @@ struct pixel_shader_uid_data
u32 early_ztest : 1;
u32 bounding_box : 1;
- // TODO: 31 bits of padding is a waste. Can we free up some bits elseware?
+ // TODO: 29 bits of padding is a waste. Can we free up some bits elseware?
u32 zfreeze : 1;
- u32 pad : 31;
+ u32 msaa : 1;
+ u32 ssaa : 1;
+ u32 pad : 29;
u32 texMtxInfo_n_projection : 8; // 8x1 bit
u32 tevindref_bi0 : 3;
diff --git a/Source/Core/VideoCommon/PixelShaderManager.h b/Source/Core/VideoCommon/PixelShaderManager.h
index 149b9cc55b..b3a7e8419b 100644
--- a/Source/Core/VideoCommon/PixelShaderManager.h
+++ b/Source/Core/VideoCommon/PixelShaderManager.h
@@ -39,7 +39,6 @@ public:
static void SetEfbScaleChanged();
static void SetZSlope(float dfdx, float dfdy, float f0);
static void SetIndMatrixChanged(int matrixidx);
- static void SetTevKSelChanged(int id);
static void SetZTextureTypeChanged();
static void SetIndTexScaleChanged(bool high);
static void SetTexCoordChanged(u8 texmapid);
diff --git a/Source/Core/VideoCommon/RenderBase.cpp b/Source/Core/VideoCommon/RenderBase.cpp
index a688c86792..55a6cb8760 100644
--- a/Source/Core/VideoCommon/RenderBase.cpp
+++ b/Source/Core/VideoCommon/RenderBase.cpp
@@ -28,6 +28,8 @@
#include "Core/Movie.h"
#include "Core/FifoPlayer/FifoRecorder.h"
+#include "Core/HW/VideoInterface.h"
+
#include "VideoCommon/AVIDump.h"
#include "VideoCommon/BPMemory.h"
#include "VideoCommon/CommandProcessor.h"
@@ -112,22 +114,23 @@ Renderer::~Renderer()
#endif
}
-void Renderer::RenderToXFB(u32 xfbAddr, const EFBRectangle& sourceRc, u32 fbWidth, u32 fbHeight, float Gamma)
+void Renderer::RenderToXFB(u32 xfbAddr, const EFBRectangle& sourceRc, u32 fbStride, u32 fbHeight, float Gamma)
{
CheckFifoRecording();
- if (!fbWidth || !fbHeight)
+ if (!fbStride || !fbHeight)
return;
XFBWrited = true;
if (g_ActiveConfig.bUseXFB)
{
- FramebufferManagerBase::CopyToXFB(xfbAddr, fbWidth, fbHeight, sourceRc, Gamma);
+ FramebufferManagerBase::CopyToXFB(xfbAddr, fbStride, fbHeight, sourceRc, Gamma);
}
else
{
- Swap(xfbAddr, fbWidth, fbWidth, fbHeight, sourceRc, Gamma);
+ // below div two to convert from bytes to pixels - it expects width, not stride
+ Swap(xfbAddr, fbStride/2, fbStride/2, fbHeight, sourceRc, Gamma);
}
}
@@ -359,15 +362,14 @@ void Renderer::DrawDebugText()
case ASPECT_AUTO:
ar_text = "Auto";
break;
- case ASPECT_FORCE_16_9:
- ar_text = "16:9";
- break;
- case ASPECT_FORCE_4_3:
- ar_text = "4:3";
- break;
case ASPECT_STRETCH:
ar_text = "Stretch";
break;
+ case ASPECT_ANALOG:
+ ar_text = "Force 4:3";
+ break;
+ case ASPECT_ANALOG_WIDE:
+ ar_text = "Force 16:9";
}
const char* const efbcopy_text = g_ActiveConfig.bSkipEFBCopyToRam ? "to Texture" : "to RAM";
@@ -400,7 +402,7 @@ void Renderer::DrawDebugText()
}
}
- final_cyan += Profiler::ToString();
+ final_cyan += Common::Profiler::ToString();
if (g_ActiveConfig.bOverlayStats)
final_cyan += Statistics::ToString();
@@ -424,31 +426,27 @@ void Renderer::UpdateDrawRectangle(int backbuffer_width, int backbuffer_height)
const float WinWidth = FloatGLWidth;
const float WinHeight = FloatGLHeight;
- // Handle aspect ratio.
- // Default to auto.
- bool use16_9 = g_aspect_wide;
-
// Update aspect ratio hack values
// Won't take effect until next frame
// Don't know if there is a better place for this code so there isn't a 1 frame delay
if (g_ActiveConfig.bWidescreenHack)
{
- float source_aspect = use16_9 ? (16.0f / 9.0f) : (4.0f / 3.0f);
+ float source_aspect = VideoInterface::GetAspectRatio(g_aspect_wide);
float target_aspect;
switch (g_ActiveConfig.iAspectRatio)
{
- case ASPECT_FORCE_16_9:
- target_aspect = 16.0f / 9.0f;
- break;
- case ASPECT_FORCE_4_3:
- target_aspect = 4.0f / 3.0f;
- break;
case ASPECT_STRETCH:
target_aspect = WinWidth / WinHeight;
break;
+ case ASPECT_ANALOG:
+ target_aspect = VideoInterface::GetAspectRatio(false);
+ break;
+ case ASPECT_ANALOG_WIDE:
+ target_aspect = VideoInterface::GetAspectRatio(true);
+ break;
default:
- // ASPECT_AUTO == no hacking
+ // ASPECT_AUTO
target_aspect = source_aspect;
break;
}
@@ -475,16 +473,24 @@ void Renderer::UpdateDrawRectangle(int backbuffer_width, int backbuffer_height)
}
// Check for force-settings and override.
- if (g_ActiveConfig.iAspectRatio == ASPECT_FORCE_16_9)
- use16_9 = true;
- else if (g_ActiveConfig.iAspectRatio == ASPECT_FORCE_4_3)
- use16_9 = false;
+
+ // The rendering window aspect ratio as a proportion of the 4:3 or 16:9 ratio
+ float Ratio;
+ switch (g_ActiveConfig.iAspectRatio)
+ {
+ case ASPECT_ANALOG_WIDE:
+ Ratio = (WinWidth / WinHeight) / VideoInterface::GetAspectRatio(true);
+ break;
+ case ASPECT_ANALOG:
+ Ratio = (WinWidth / WinHeight) / VideoInterface::GetAspectRatio(false);
+ break;
+ default:
+ Ratio = (WinWidth / WinHeight) / VideoInterface::GetAspectRatio(g_aspect_wide);
+ break;
+ }
if (g_ActiveConfig.iAspectRatio != ASPECT_STRETCH)
{
- // The rendering window aspect ratio as a proportion of the 4:3 or 16:9 ratio
- float Ratio = (WinWidth / WinHeight) / (!use16_9 ? (4.0f / 3.0f) : (16.0f / 9.0f));
- // Check if height or width is the limiting factor. If ratio > 1 the picture is too wide and have to limit the width.
if (Ratio > 1.0f)
{
// Scale down and center in the X direction.
@@ -501,12 +507,27 @@ void Renderer::UpdateDrawRectangle(int backbuffer_width, int backbuffer_height)
}
// -----------------------------------------------------------------------
- // Crop the picture from 4:3 to 5:4 or from 16:9 to 16:10.
+ // Crop the picture from Analog to 4:3 or from Analog (Wide) to 16:9.
// Output: FloatGLWidth, FloatGLHeight, FloatXOffset, FloatYOffset
// ------------------
if (g_ActiveConfig.iAspectRatio != ASPECT_STRETCH && g_ActiveConfig.bCrop)
{
- float Ratio = !use16_9 ? ((4.0f / 3.0f) / (5.0f / 4.0f)) : (((16.0f / 9.0f) / (16.0f / 10.0f)));
+ switch (g_ActiveConfig.iAspectRatio)
+ {
+ case ASPECT_ANALOG_WIDE:
+ Ratio = (16.0f / 9.0f) / VideoInterface::GetAspectRatio(true);
+ break;
+ case ASPECT_ANALOG:
+ Ratio = (4.0f / 3.0f) / VideoInterface::GetAspectRatio(false);
+ break;
+ default:
+ Ratio = (!g_aspect_wide ? (4.0f / 3.0f) : (16.0f / 9.0f)) / VideoInterface::GetAspectRatio(g_aspect_wide);
+ break;
+ }
+ if (Ratio <= 1.0f)
+ {
+ Ratio = 1.0f / Ratio;
+ }
// The width and height we will add (calculate this before FloatGLWidth and FloatGLHeight is adjusted)
float IncreasedWidth = (Ratio - 1.0f) * FloatGLWidth;
float IncreasedHeight = (Ratio - 1.0f) * FloatGLHeight;
diff --git a/Source/Core/VideoCommon/RenderBase.h b/Source/Core/VideoCommon/RenderBase.h
index a17525ca2e..c717483d1e 100644
--- a/Source/Core/VideoCommon/RenderBase.h
+++ b/Source/Core/VideoCommon/RenderBase.h
@@ -109,7 +109,7 @@ public:
virtual void ClearScreen(const EFBRectangle& rc, bool colorEnable, bool alphaEnable, bool zEnable, u32 color, u32 z) = 0;
virtual void ReinterpretPixelData(unsigned int convtype) = 0;
- static void RenderToXFB(u32 xfbAddr, const EFBRectangle& sourceRc, u32 fbWidth, u32 fbHeight, float Gamma = 1.0f);
+ static void RenderToXFB(u32 xfbAddr, const EFBRectangle& sourceRc, u32 fbStride, u32 fbHeight, float Gamma = 1.0f);
virtual u32 AccessEFB(EFBAccessType type, u32 x, u32 y, u32 poke_data) = 0;
virtual void PokeEFB(EFBAccessType type, const std::vector<EfbPokeData>& data);
diff --git a/Source/Core/VideoCommon/ShaderGenCommon.h b/Source/Core/VideoCommon/ShaderGenCommon.h
index 218d9b98b5..2fd6c51819 100644
--- a/Source/Core/VideoCommon/ShaderGenCommon.h
+++ b/Source/Core/VideoCommon/ShaderGenCommon.h
@@ -279,6 +279,29 @@ static inline void AssignVSOutputMembers(T& object, const char* a, const char* b
}
}
+// We use the flag "centroid" to fix some MSAA rendering bugs. With MSAA, the
+// pixel shader will be executed for each pixel which has at least one passed sample.
+// So there may be rendered pixels where the center of the pixel isn't in the primitive.
+// As the pixel shader usually renders at the center of the pixel, this position may be
+// outside the primitive. This will lead to sampling outside the texture, sign changes, ...
+// As a workaround, we interpolate at the centroid of the coveraged pixel, which
+// is always inside the primitive.
+// Without MSAA, this flag is defined to have no effect.
+static inline const char* GetInterpolationQualifier(API_TYPE api_type, bool in = true, bool in_out = false)
+{
+ if (!g_ActiveConfig.iMultisampleMode)
+ return "";
+
+ if (!g_ActiveConfig.bSSAA)
+ {
+ if (in_out && api_type == API_OPENGL && !g_ActiveConfig.backend_info.bSupportsBindingLayout)
+ return in ? "centroid in" : "centroid out";
+ return "centroid";
+ }
+
+ return "sample";
+}
+
// Constant variable names
#define I_COLORS "color"
#define I_KCOLORS "k"
diff --git a/Source/Core/VideoCommon/TextureCacheBase.cpp b/Source/Core/VideoCommon/TextureCacheBase.cpp
index 0b0588cf6e..1d9b564a8f 100644
--- a/Source/Core/VideoCommon/TextureCacheBase.cpp
+++ b/Source/Core/VideoCommon/TextureCacheBase.cpp
@@ -10,6 +10,8 @@
#include "Common/StringUtil.h"
#include "Core/ConfigManager.h"
+#include "Core/FifoPlayer/FifoPlayer.h"
+#include "Core/FifoPlayer/FifoRecorder.h"
#include "Core/HW/Memmap.h"
#include "VideoCommon/Debugger.h"
@@ -18,16 +20,17 @@
#include "VideoCommon/RenderBase.h"
#include "VideoCommon/Statistics.h"
#include "VideoCommon/TextureCacheBase.h"
+#include "VideoCommon/VideoCommon.h"
#include "VideoCommon/VideoConfig.h"
static const u64 TEXHASH_INVALID = 0;
-static const int TEXTURE_KILL_THRESHOLD = 60;
+static const int TEXTURE_KILL_THRESHOLD = 64; // Sonic the Fighters (inside Sonic Gems Collection) loops a 64 frames animation
static const int TEXTURE_POOL_KILL_THRESHOLD = 3;
static const int FRAMECOUNT_INVALID = 0;
-TextureCache *g_texture_cache;
+TextureCache* g_texture_cache;
-GC_ALIGNED16(u8 *TextureCache::temp) = nullptr;
+alignas(16) u8* TextureCache::temp = nullptr;
size_t TextureCache::temp_size;
TextureCache::TexCache TextureCache::textures_by_address;
@@ -149,12 +152,28 @@ void TextureCache::Cleanup(int _frameCount)
if (iter->second->frameCount == FRAMECOUNT_INVALID)
{
iter->second->frameCount = _frameCount;
+ ++iter;
}
- if (_frameCount > TEXTURE_KILL_THRESHOLD + iter->second->frameCount &&
- // EFB copies living on the host GPU are unrecoverable and thus shouldn't be deleted
- !iter->second->IsEfbCopy())
+ else if (_frameCount > TEXTURE_KILL_THRESHOLD + iter->second->frameCount)
{
- iter = FreeTexture(iter);
+ if (iter->second->IsEfbCopy())
+ {
+ // Only remove EFB copies when they wouldn't be used anymore(changed hash), because EFB copies living on the
+ // host GPU are unrecoverable. Perform this check only every TEXTURE_KILL_THRESHOLD for performance reasons
+ if ((_frameCount - iter->second->frameCount) % TEXTURE_KILL_THRESHOLD == 1 &&
+ iter->second->hash != iter->second->CalculateHash())
+ {
+ iter = FreeTexture(iter);
+ }
+ else
+ {
+ ++iter;
+ }
+ }
+ else
+ {
+ iter = FreeTexture(iter);
+ }
}
else
{
@@ -182,24 +201,6 @@ void TextureCache::Cleanup(int _frameCount)
}
}
-void TextureCache::MakeRangeDynamic(u32 start_address, u32 size)
-{
- TexCache::iterator
- iter = textures_by_address.begin();
-
- while (iter != textures_by_address.end())
- {
- if (iter->second->OverlapsMemoryRange(start_address, size))
- {
- iter = FreeTexture(iter);
- }
- else
- {
- ++iter;
- }
- }
-}
-
bool TextureCache::TCacheEntryBase::OverlapsMemoryRange(u32 range_address, u32 range_size) const
{
if (addr + size_in_bytes <= range_address)
@@ -211,46 +212,98 @@ bool TextureCache::TCacheEntryBase::OverlapsMemoryRange(u32 range_address, u32 r
return true;
}
-void TextureCache::TCacheEntryBase::DoPartialTextureUpdates()
+TextureCache::TCacheEntryBase* TextureCache::DoPartialTextureUpdates(TexCache::iterator iter_t)
{
- const bool isPaletteTexture = (format== GX_TF_C4 || format == GX_TF_C8 || format == GX_TF_C14X2 || format >= 0x10000);
+ TCacheEntryBase* entry_to_update = iter_t->second;
+ const bool isPaletteTexture = (entry_to_update->format == GX_TF_C4
+ || entry_to_update->format == GX_TF_C8
+ || entry_to_update->format == GX_TF_C14X2
+ || entry_to_update->format >= 0x10000);
// Efb copies and paletted textures are excluded from these updates, until there's an example where a game would
// benefit from this. Both would require more work to be done.
// TODO: Implement upscaling support for normal textures, and then remove the efb to ram and the scaled efb restrictions
- if (!g_ActiveConfig.backend_info.bSupportsCopySubImage || !g_ActiveConfig.bSkipEFBCopyToRam || IsEfbCopy()
- || isPaletteTexture || (g_ActiveConfig.bCopyEFBScaled && g_ActiveConfig.iEFBScale != SCALE_1X))
- return;
+ if (entry_to_update->IsEfbCopy()
+ || isPaletteTexture)
+ return entry_to_update;
- u32 block_width = TexDecoder_GetBlockWidthInTexels(format);
- u32 block_height = TexDecoder_GetBlockHeightInTexels(format);
- u32 block_size = block_width * block_height * TexDecoder_GetTexelSizeInNibbles(format) / 2;
+ u32 block_width = TexDecoder_GetBlockWidthInTexels(entry_to_update->format);
+ u32 block_height = TexDecoder_GetBlockHeightInTexels(entry_to_update->format);
+ u32 block_size = block_width * block_height * TexDecoder_GetTexelSizeInNibbles(entry_to_update->format) / 2;
- u32 numBlocksX = (native_width + block_width - 1) / block_width;
-
- TexCache::iterator iter = textures_by_address.lower_bound(addr);
- TexCache::iterator iterend = textures_by_address.upper_bound(addr + size_in_bytes);
+ u32 numBlocksX = (entry_to_update->native_width + block_width - 1) / block_width;
+ TexCache::iterator iter = textures_by_address.lower_bound(entry_to_update->addr);
+ TexCache::iterator iterend = textures_by_address.upper_bound(entry_to_update->addr + entry_to_update->size_in_bytes);
+ bool entry_need_scaling = true;
while (iter != iterend)
{
TCacheEntryBase* entry = iter->second;
- if (entry->IsEfbCopy() && addr <= entry->addr && entry->addr + entry->size_in_bytes <= addr + size_in_bytes
- && entry->frameCount == FRAMECOUNT_INVALID && entry->copyMipMapStrideChannels * 32 == numBlocksX * block_size)
+ if (entry != entry_to_update
+ && entry->IsEfbCopy()
+ && entry_to_update->addr <= entry->addr
+ && entry->addr + entry->size_in_bytes <= entry_to_update->addr + entry_to_update->size_in_bytes
+ && entry->frameCount == FRAMECOUNT_INVALID
+ && entry->memory_stride == numBlocksX * block_size)
{
- u32 block_offset = (entry->addr - addr) / block_size;
+ u32 block_offset = (entry->addr - entry_to_update->addr) / block_size;
u32 block_x = block_offset % numBlocksX;
u32 block_y = block_offset / numBlocksX;
u32 x = block_x * block_width;
u32 y = block_y * block_height;
-
- DoPartialTextureUpdate(entry, x, y);
-
+ MathUtil::Rectangle<int> srcrect, dstrect;
+ srcrect.left = 0;
+ srcrect.top = 0;
+ dstrect.left = 0;
+ dstrect.top = 0;
+ if (entry_need_scaling)
+ {
+ entry_need_scaling = false;
+ u32 w = entry_to_update->native_width * entry->config.width / entry->native_width;
+ u32 h = entry_to_update->native_height * entry->config.height / entry->native_height;
+ u32 max = g_renderer->GetMaxTextureSize();
+ if (max < w || max < h)
+ {
+ iter++;
+ continue;
+ }
+ if (entry_to_update->config.width != w || entry_to_update->config.height != h)
+ {
+ TextureCache::TCacheEntryConfig newconfig;
+ newconfig.width = w;
+ newconfig.height = h;
+ newconfig.rendertarget = true;
+ TCacheEntryBase* newentry = AllocateTexture(newconfig);
+ newentry->SetGeneralParameters(entry_to_update->addr, entry_to_update->size_in_bytes, entry_to_update->format);
+ newentry->SetDimensions(entry_to_update->native_width, entry_to_update->native_height, 1);
+ newentry->SetHashes(entry_to_update->base_hash, entry_to_update->hash);
+ newentry->frameCount = frameCount;
+ newentry->is_efb_copy = false;
+ srcrect.right = entry_to_update->config.width;
+ srcrect.bottom = entry_to_update->config.height;
+ dstrect.right = w;
+ dstrect.bottom = h;
+ newentry->CopyRectangleFromTexture(entry_to_update, srcrect, dstrect);
+ entry_to_update = newentry;
+ u64 key = iter_t->first;
+ iter_t = FreeTexture(iter_t);
+ textures_by_address.emplace(key, entry_to_update);
+ }
+ }
+ srcrect.right = entry->config.width;
+ srcrect.bottom = entry->config.height;
+ dstrect.left = x * entry_to_update->config.width / entry_to_update->native_width;
+ dstrect.top = y * entry_to_update->config.height / entry_to_update->native_height;
+ dstrect.right = (x + entry->native_width) * entry_to_update->config.width / entry_to_update->native_width;
+ dstrect.bottom = (y + entry->native_height) * entry_to_update->config.height / entry_to_update->native_height;
+ entry_to_update->CopyRectangleFromTexture(entry, srcrect, dstrect);
// Mark the texture update as used, so it isn't applied more than once
entry->frameCount = frameCount;
}
++iter;
}
+ return entry_to_update;
}
void TextureCache::DumpTexture(TCacheEntryBase* entry, std::string basename, unsigned int level)
@@ -320,16 +373,16 @@ TextureCache::TCacheEntryBase* TextureCache::Load(const u32 stage)
return nullptr;
// TexelSizeInNibbles(format) * width * height / 16;
- const unsigned int bsw = TexDecoder_GetBlockWidthInTexels(texformat) - 1;
- const unsigned int bsh = TexDecoder_GetBlockHeightInTexels(texformat) - 1;
+ const unsigned int bsw = TexDecoder_GetBlockWidthInTexels(texformat);
+ const unsigned int bsh = TexDecoder_GetBlockHeightInTexels(texformat);
- unsigned int expandedWidth = (width + bsw) & (~bsw);
- unsigned int expandedHeight = (height + bsh) & (~bsh);
+ unsigned int expandedWidth = ROUND_UP(width, bsw);
+ unsigned int expandedHeight = ROUND_UP(height, bsh);
const unsigned int nativeW = width;
const unsigned int nativeH = height;
// Hash assigned to texcache entry (also used to generate filenames used for texture dumping and custom texture lookup)
- u64 tex_hash = TEXHASH_INVALID;
+ u64 base_hash = TEXHASH_INVALID;
u64 full_hash = TEXHASH_INVALID;
u32 full_format = texformat;
@@ -344,6 +397,25 @@ TextureCache::TCacheEntryBase* TextureCache::Load(const u32 stage)
full_format = texformat | (tlutfmt << 16);
const u32 texture_size = TexDecoder_GetTextureSizeInBytes(expandedWidth, expandedHeight, texformat);
+ u32 additional_mips_size = 0; // not including level 0, which is texture_size
+
+ // GPUs don't like when the specified mipmap count would require more than one 1x1-sized LOD in the mipmap chain
+ // e.g. 64x64 with 7 LODs would have the mipmap chain 64x64,32x32,16x16,8x8,4x4,2x2,1x1,0x0, so we limit the mipmap count to 6 there
+ tex_levels = std::min<u32>(IntLog2(std::max(width, height)) + 1, tex_levels);
+
+ for (u32 level = 1; level != tex_levels; ++level)
+ {
+ // We still need to calculate the original size of the mips
+ const u32 expanded_mip_width = ROUND_UP(CalculateLevelSize(width, level), bsw);
+ const u32 expanded_mip_height = ROUND_UP(CalculateLevelSize(height, level), bsh);
+
+ additional_mips_size += TexDecoder_GetTextureSizeInBytes(expanded_mip_width, expanded_mip_height, texformat);
+ }
+
+ // If we are recording a FifoLog, keep track of what memory we read.
+ // FifiRecorder does it's own memory modification tracking independant of the texture hashing below.
+ if (g_bRecordFifoData && !from_tmem)
+ FifoRecorder::GetInstance().UseMemory(address, texture_size + additional_mips_size, MemoryUpdate::TEXTURE_MAP);
const u8* src_data;
if (from_tmem)
@@ -352,22 +424,18 @@ TextureCache::TCacheEntryBase* TextureCache::Load(const u32 stage)
src_data = Memory::GetPointer(address);
// TODO: This doesn't hash GB tiles for preloaded RGBA8 textures (instead, it's hashing more data from the low tmem bank than it should)
- tex_hash = GetHash64(src_data, texture_size, g_ActiveConfig.iSafeTextureCache_ColorSamples);
+ base_hash = GetHash64(src_data, texture_size, g_ActiveConfig.iSafeTextureCache_ColorSamples);
u32 palette_size = 0;
if (isPaletteTexture)
{
palette_size = TexDecoder_GetPaletteSize(texformat);
- full_hash = tex_hash ^ GetHash64(&texMem[tlutaddr], palette_size, g_ActiveConfig.iSafeTextureCache_ColorSamples);
+ full_hash = base_hash ^ GetHash64(&texMem[tlutaddr], palette_size, g_ActiveConfig.iSafeTextureCache_ColorSamples);
}
else
{
- full_hash = tex_hash;
+ full_hash = base_hash;
}
- // GPUs don't like when the specified mipmap count would require more than one 1x1-sized LOD in the mipmap chain
- // e.g. 64x64 with 7 LODs would have the mipmap chain 64x64,32x32,16x16,8x8,4x4,2x2,1x1,0x0, so we limit the mipmap count to 6 there
- tex_levels = std::min<u32>(IntLog2(std::max(width, height)) + 1, tex_levels);
-
// Search the texture cache for textures by address
//
// Find all texture cache entries for the current texture address, and decide whether to use one of
@@ -405,11 +473,10 @@ TextureCache::TCacheEntryBase* TextureCache::Load(const u32 stage)
TCacheEntryBase* entry = iter->second;
if (entry->IsEfbCopy())
{
- // EFB copies have slightly different rules: the hash doesn't need to match
- // in EFB2Tex mode, and EFB copy formats have different meanings from texture
- // formats.
- if (g_ActiveConfig.bSkipEFBCopyToRam ||
- (tex_hash == entry->hash && (!isPaletteTexture || g_Config.backend_info.bSupportsPaletteConversion)))
+ // EFB copies have slightly different rules as EFB copy formats have different
+ // meanings from texture formats.
+ if ((base_hash == entry->hash && (!isPaletteTexture || g_Config.backend_info.bSupportsPaletteConversion)) ||
+ IsPlayingBackFifologWithBrokenEFBCopies)
{
// TODO: We should check format/width/height/levels for EFB copies. Checking
// format is complicated because EFB copy formats don't exactly match
@@ -440,14 +507,18 @@ TextureCache::TCacheEntryBase* TextureCache::Load(const u32 stage)
if (entry->hash == full_hash && entry->format == full_format && entry->native_levels >= tex_levels &&
entry->native_width == nativeW && entry->native_height == nativeH)
{
- entry->DoPartialTextureUpdates();
+ entry = DoPartialTextureUpdates(iter);
return ReturnEntry(stage, entry);
}
}
- // Find the entry which hasn't been used for the longest time
- if (entry->frameCount != FRAMECOUNT_INVALID && entry->frameCount < temp_frameCount)
+ // Find the texture which hasn't been used for the longest time. Count paletted
+ // textures as the same texture here, when the texture itself is the same. This
+ // improves the performance a lot in some games that use paletted textures.
+ // Example: Sonic the Fighters (inside Sonic Gems Collection)
+ if (entry->frameCount != FRAMECOUNT_INVALID && entry->frameCount < temp_frameCount &&
+ !(isPaletteTexture && entry->base_hash == base_hash))
{
temp_frameCount = entry->frameCount;
oldest_entry = iter;
@@ -469,12 +540,12 @@ TextureCache::TCacheEntryBase* TextureCache::Load(const u32 stage)
decoded_entry->SetGeneralParameters(address, texture_size, full_format);
decoded_entry->SetDimensions(entry->native_width, entry->native_height, 1);
- decoded_entry->SetHashes(full_hash);
+ decoded_entry->SetHashes(base_hash, full_hash);
decoded_entry->frameCount = FRAMECOUNT_INVALID;
decoded_entry->is_efb_copy = false;
g_texture_cache->ConvertTexture(decoded_entry, entry, &texMem[tlutaddr], (TlutFormat)tlutfmt);
- textures_by_address.insert(TexCache::value_type((u64)address, decoded_entry));
+ textures_by_address.emplace((u64)address, decoded_entry);
return ReturnEntry(stage, decoded_entry);
}
@@ -494,7 +565,7 @@ TextureCache::TCacheEntryBase* TextureCache::Load(const u32 stage)
if (entry->format == full_format && entry->native_levels >= tex_levels &&
entry->native_width == nativeW && entry->native_height == nativeH)
{
- entry->DoPartialTextureUpdates();
+ entry = DoPartialTextureUpdates(iter);
return ReturnEntry(stage, entry);
}
@@ -539,7 +610,7 @@ TextureCache::TCacheEntryBase* TextureCache::Load(const u32 stage)
if (!(texformat == GX_TF_RGBA8 && from_tmem))
{
const u8* tlut = &texMem[tlutaddr];
- TexDecoder_Decode(temp, src_data, expandedWidth, expandedHeight, texformat, tlut, (TlutFormat) tlutfmt);
+ TexDecoder_Decode(temp, src_data, expandedWidth, expandedHeight, texformat, tlut, (TlutFormat)tlutfmt);
}
else
{
@@ -560,16 +631,16 @@ TextureCache::TCacheEntryBase* TextureCache::Load(const u32 stage)
TCacheEntryBase* entry = AllocateTexture(config);
GFX_DEBUGGER_PAUSE_AT(NEXT_NEW_TEXTURE, true);
- textures_by_address.insert(TexCache::value_type((u64)address, entry));
+ iter = textures_by_address.emplace((u64)address, entry);
if (g_ActiveConfig.iSafeTextureCache_ColorSamples == 0 ||
std::max(texture_size, palette_size) <= (u32)g_ActiveConfig.iSafeTextureCache_ColorSamples * 8)
{
- entry->textures_by_hash_iter = textures_by_hash.insert(TexCache::value_type(full_hash, entry));
+ entry->textures_by_hash_iter = textures_by_hash.emplace(full_hash, entry);
}
entry->SetGeneralParameters(address, texture_size, full_format);
entry->SetDimensions(nativeW, nativeH, tex_levels);
- entry->hash = full_hash;
+ entry->SetHashes(base_hash, full_hash);
entry->is_efb_copy = false;
entry->is_custom_tex = hires_tex != nullptr;
@@ -616,8 +687,8 @@ TextureCache::TCacheEntryBase* TextureCache::Load(const u32 stage)
{
const u32 mip_width = CalculateLevelSize(width, level);
const u32 mip_height = CalculateLevelSize(height, level);
- const u32 expanded_mip_width = (mip_width + bsw) & (~bsw);
- const u32 expanded_mip_height = (mip_height + bsh) & (~bsh);
+ const u32 expanded_mip_width = ROUND_UP(mip_width, bsw);
+ const u32 expanded_mip_height = ROUND_UP(mip_height, bsh);
const u8*& mip_src_data = from_tmem
? ((level % 2) ? ptr_odd : ptr_even)
@@ -636,12 +707,12 @@ TextureCache::TCacheEntryBase* TextureCache::Load(const u32 stage)
INCSTAT(stats.numTexturesUploaded);
SETSTAT(stats.numTexturesAlive, textures_by_address.size());
- entry->DoPartialTextureUpdates();
+ entry = DoPartialTextureUpdates(iter);
return ReturnEntry(stage, entry);
}
-void TextureCache::CopyRenderTargetToTexture(u32 dstAddr, unsigned int dstFormat, PEControl::PixelFormat srcFormat,
+void TextureCache::CopyRenderTargetToTexture(u32 dstAddr, unsigned int dstFormat, u32 dstStride, PEControl::PixelFormat srcFormat,
const EFBRectangle& srcRect, bool isIntensity, bool scaleByHalf)
{
// Emulation methods:
@@ -686,7 +757,7 @@ void TextureCache::CopyRenderTargetToTexture(u32 dstAddr, unsigned int dstFormat
//
// For historical reasons, Dolphin doesn't actually implement "pure" EFB to RAM emulation, but only EFB to texture and hybrid EFB copies.
- float colmat[28] = {0};
+ float colmat[28] = { 0 };
float *const fConstAdd = colmat + 16;
float *const ColorMask = colmat + 20;
ColorMask[0] = ColorMask[1] = ColorMask[2] = ColorMask[3] = 255.0f;
@@ -701,9 +772,11 @@ void TextureCache::CopyRenderTargetToTexture(u32 dstAddr, unsigned int dstFormat
case 0: // Z4
colmat[3] = colmat[7] = colmat[11] = colmat[15] = 1.0f;
cbufid = 0;
+ dstFormat |= _GX_TF_CTF;
break;
+ case 8: // Z8H
+ dstFormat |= _GX_TF_CTF;
case 1: // Z8
- case 8: // Z8
colmat[0] = colmat[4] = colmat[8] = colmat[12] = 1.0f;
cbufid = 1;
break;
@@ -716,6 +789,7 @@ void TextureCache::CopyRenderTargetToTexture(u32 dstAddr, unsigned int dstFormat
case 11: // Z16 (reverse order)
colmat[0] = colmat[4] = colmat[8] = colmat[13] = 1.0f;
cbufid = 3;
+ dstFormat |= _GX_TF_CTF;
break;
case 6: // Z24X8
@@ -726,11 +800,13 @@ void TextureCache::CopyRenderTargetToTexture(u32 dstAddr, unsigned int dstFormat
case 9: // Z8M
colmat[1] = colmat[5] = colmat[9] = colmat[13] = 1.0f;
cbufid = 5;
+ dstFormat |= _GX_TF_CTF;
break;
case 10: // Z8L
colmat[2] = colmat[6] = colmat[10] = colmat[14] = 1.0f;
cbufid = 6;
+ dstFormat |= _GX_TF_CTF;
break;
case 12: // Z16L - copy lower 16 depth bits
@@ -738,6 +814,7 @@ void TextureCache::CopyRenderTargetToTexture(u32 dstAddr, unsigned int dstFormat
// Used e.g. in Zelda: Skyward Sword
colmat[1] = colmat[5] = colmat[9] = colmat[14] = 1.0f;
cbufid = 7;
+ dstFormat |= _GX_TF_CTF;
break;
default:
@@ -746,6 +823,8 @@ void TextureCache::CopyRenderTargetToTexture(u32 dstAddr, unsigned int dstFormat
cbufid = 8;
break;
}
+
+ dstFormat |= _GX_TF_ZTF;
}
else if (isIntensity)
{
@@ -810,11 +889,13 @@ void TextureCache::CopyRenderTargetToTexture(u32 dstAddr, unsigned int dstFormat
ColorMask[0] = 15.0f;
ColorMask[4] = 1.0f / 15.0f;
cbufid = 14;
+ dstFormat |= _GX_TF_CTF;
break;
case 1: // R8
case 8: // R8
colmat[0] = colmat[4] = colmat[8] = colmat[12] = 1;
cbufid = 15;
+ dstFormat |= _GX_TF_CTF;
break;
case 2: // RA4
@@ -829,6 +910,7 @@ void TextureCache::CopyRenderTargetToTexture(u32 dstAddr, unsigned int dstFormat
fConstAdd[3] = 1.0f;
cbufid = 17;
}
+ dstFormat |= _GX_TF_CTF;
break;
case 3: // RA8
colmat[0] = colmat[4] = colmat[8] = colmat[15] = 1.0f;
@@ -840,6 +922,7 @@ void TextureCache::CopyRenderTargetToTexture(u32 dstAddr, unsigned int dstFormat
fConstAdd[3] = 1.0f;
cbufid = 19;
}
+ dstFormat |= _GX_TF_CTF;
break;
case 7: // A8
@@ -855,25 +938,30 @@ void TextureCache::CopyRenderTargetToTexture(u32 dstAddr, unsigned int dstFormat
fConstAdd[3] = 1.0f;
cbufid = 21;
}
+ dstFormat |= _GX_TF_CTF;
break;
case 9: // G8
colmat[1] = colmat[5] = colmat[9] = colmat[13] = 1.0f;
cbufid = 22;
+ dstFormat |= _GX_TF_CTF;
break;
case 10: // B8
colmat[2] = colmat[6] = colmat[10] = colmat[14] = 1.0f;
cbufid = 23;
+ dstFormat |= _GX_TF_CTF;
break;
case 11: // RG8
colmat[0] = colmat[4] = colmat[8] = colmat[13] = 1.0f;
cbufid = 24;
+ dstFormat |= _GX_TF_CTF;
break;
case 12: // GB8
colmat[1] = colmat[5] = colmat[9] = colmat[14] = 1.0f;
cbufid = 25;
+ dstFormat |= _GX_TF_CTF;
break;
case 4: // RGB565
@@ -921,6 +1009,13 @@ void TextureCache::CopyRenderTargetToTexture(u32 dstAddr, unsigned int dstFormat
}
}
+ u8* dst = Memory::GetPointer(dstAddr);
+ if (dst == nullptr)
+ {
+ ERROR_LOG(VIDEO, "Trying to copy from EFB to invalid address 0x%8x", dstAddr);
+ return;
+ }
+
const unsigned int tex_w = scaleByHalf ? srcRect.GetWidth() / 2 : srcRect.GetWidth();
const unsigned int tex_h = scaleByHalf ? srcRect.GetHeight() / 2 : srcRect.GetHeight();
@@ -928,11 +1023,13 @@ void TextureCache::CopyRenderTargetToTexture(u32 dstAddr, unsigned int dstFormat
unsigned int scaled_tex_h = g_ActiveConfig.bCopyEFBScaled ? Renderer::EFBToScaledY(tex_h) : tex_h;
// remove all texture cache entries at dstAddr
- std::pair<TexCache::iterator, TexCache::iterator> iter_range = textures_by_address.equal_range((u64)dstAddr);
- TexCache::iterator iter = iter_range.first;
- while (iter != iter_range.second)
{
- iter = FreeTexture(iter);
+ std::pair<TexCache::iterator, TexCache::iterator> iter_range = textures_by_address.equal_range((u64)dstAddr);
+ TexCache::iterator iter = iter_range.first;
+ while (iter != iter_range.second)
+ {
+ iter = FreeTexture(iter);
+ }
}
// create the texture
@@ -944,17 +1041,32 @@ void TextureCache::CopyRenderTargetToTexture(u32 dstAddr, unsigned int dstFormat
TCacheEntryBase* entry = AllocateTexture(config);
- // TODO: Using the wrong dstFormat, dumb...
entry->SetGeneralParameters(dstAddr, 0, dstFormat);
entry->SetDimensions(tex_w, tex_h, 1);
- entry->SetHashes(TEXHASH_INVALID);
entry->frameCount = FRAMECOUNT_INVALID;
- entry->is_efb_copy = true;
+ entry->SetEfbCopy(dstStride);
entry->is_custom_tex = false;
- entry->copyMipMapStrideChannels = bpmem.copyMipMapStrideChannels;
- entry->FromRenderTarget(dstAddr, dstFormat, srcFormat, srcRect, isIntensity, scaleByHalf, cbufid, colmat);
+ entry->FromRenderTarget(dst, dstFormat, dstStride, srcFormat, srcRect, isIntensity, scaleByHalf, cbufid, colmat);
+
+ u64 hash = entry->CalculateHash();
+ entry->SetHashes(hash, hash);
+
+ // Invalidate all textures that overlap the range of our efb copy.
+ // Unless our efb copy has a weird stride, then we want avoid invalidating textures which
+ // we might be able to do a partial texture update on.
+ if (entry->memory_stride == entry->CacheLinesPerRow() * 32)
+ {
+ TexCache::iterator iter = textures_by_address.begin();
+ while (iter != textures_by_address.end())
+ {
+ if (iter->second->OverlapsMemoryRange(dstAddr, entry->size_in_bytes))
+ iter = FreeTexture(iter);
+ else
+ ++iter;
+ }
+ }
if (g_ActiveConfig.bDumpEFBTarget)
{
@@ -963,7 +1075,18 @@ void TextureCache::CopyRenderTargetToTexture(u32 dstAddr, unsigned int dstFormat
count++), 0);
}
- textures_by_address.insert(TexCache::value_type((u64)dstAddr, entry));
+ if (g_bRecordFifoData)
+ {
+ // Mark the memory behind this efb copy as dynamicly generated for the Fifo log
+ u32 address = dstAddr;
+ for (u32 i = 0; i < entry->NumBlocksY(); i++)
+ {
+ FifoRecorder::GetInstance().UseMemory(address, entry->CacheLinesPerRow() * 32, MemoryUpdate::TEXTURE_MAP, true);
+ address += entry->memory_stride;
+ }
+ }
+
+ textures_by_address.emplace((u64)dstAddr, entry);
}
TextureCache::TCacheEntryBase* TextureCache::AllocateTexture(const TCacheEntryConfig& config)
@@ -996,7 +1119,79 @@ TextureCache::TexCache::iterator TextureCache::FreeTexture(TexCache::iterator it
}
entry->frameCount = FRAMECOUNT_INVALID;
- texture_pool.insert(TexPool::value_type(entry->config, entry));
+ texture_pool.emplace(entry->config, entry);
return textures_by_address.erase(iter);
}
+
+u32 TextureCache::TCacheEntryBase::CacheLinesPerRow() const
+{
+ u32 blockW = TexDecoder_GetBlockWidthInTexels(format);
+ // Round up source height to multiple of block size
+ u32 actualWidth = ROUND_UP(native_width, blockW);
+
+ u32 numBlocksX = actualWidth / blockW;
+
+ // RGBA takes two cache lines per block; all others take one
+ if (format == GX_TF_RGBA8)
+ numBlocksX = numBlocksX * 2;
+ return numBlocksX;
+}
+
+u32 TextureCache::TCacheEntryBase::NumBlocksY() const
+{
+ u32 blockH = TexDecoder_GetBlockHeightInTexels(format);
+ // Round up source height to multiple of block size
+ u32 actualHeight = ROUND_UP(native_height, blockH);
+
+ return actualHeight / blockH;
+}
+
+void TextureCache::TCacheEntryBase::SetEfbCopy(u32 stride)
+{
+ is_efb_copy = true;
+ memory_stride = stride;
+
+ _assert_msg_(VIDEO, memory_stride >= CacheLinesPerRow(), "Memory stride is too small");
+
+ size_in_bytes = memory_stride * NumBlocksY();
+}
+
+// Fill gamecube memory backing this texture with zeros.
+void TextureCache::TCacheEntryBase::Zero(u8* ptr)
+{
+ for (u32 i = 0; i < NumBlocksY(); i++)
+ {
+ memset(ptr, 0, CacheLinesPerRow() * 32);
+ ptr += memory_stride;
+ }
+}
+
+u64 TextureCache::TCacheEntryBase::CalculateHash() const
+{
+ u8* ptr = Memory::GetPointer(addr);
+ if (memory_stride == CacheLinesPerRow() * 32)
+ {
+ return GetHash64(ptr, size_in_bytes, g_ActiveConfig.iSafeTextureCache_ColorSamples);
+ }
+ else
+ {
+ u32 blocks = NumBlocksY();
+ u64 temp_hash = size_in_bytes;
+
+ u32 samples_per_row = 0;
+ if (g_ActiveConfig.iSafeTextureCache_ColorSamples != 0)
+ {
+ // Hash at least 4 samples per row to avoid hashing in a bad pattern, like just on the left side of the efb copy
+ samples_per_row = std::max(g_ActiveConfig.iSafeTextureCache_ColorSamples / blocks, 4u);
+ }
+
+ for (u32 i = 0; i < blocks; i++)
+ {
+ // Multiply by a prime number to mix the hash up a bit. This prevents identical blocks from canceling each other out
+ temp_hash = (temp_hash * 397) ^ GetHash64(ptr, CacheLinesPerRow() * 32, samples_per_row);
+ ptr += memory_stride;
+ }
+ return temp_hash;
+ }
+}
diff --git a/Source/Core/VideoCommon/TextureCacheBase.h b/Source/Core/VideoCommon/TextureCacheBase.h
index 5d4f9204fc..4fbaed8e15 100644
--- a/Source/Core/VideoCommon/TextureCacheBase.h
+++ b/Source/Core/VideoCommon/TextureCacheBase.h
@@ -49,11 +49,12 @@ public:
// common members
u32 addr;
u32 size_in_bytes;
- u64 hash;
+ u64 base_hash;
+ u64 hash; // for paletted textures, hash = base_hash ^ palette_hash
u32 format;
bool is_efb_copy;
bool is_custom_tex;
- u32 copyMipMapStrideChannels;
+ u32 memory_stride;
unsigned int native_width, native_height; // Texture dimensions from the GameCube's point of view
unsigned int native_levels;
@@ -76,33 +77,45 @@ public:
native_width = _native_width;
native_height = _native_height;
native_levels = _native_levels;
+ memory_stride = _native_width;
}
- void SetHashes(u64 _hash)
+ void SetHashes(u64 _base_hash, u64 _hash)
{
+ base_hash = _base_hash;
hash = _hash;
}
+ void SetEfbCopy(u32 stride);
+
TCacheEntryBase(const TCacheEntryConfig& c) : config(c) {}
virtual ~TCacheEntryBase();
virtual void Bind(unsigned int stage) = 0;
virtual bool Save(const std::string& filename, unsigned int level) = 0;
- virtual void DoPartialTextureUpdate(TCacheEntryBase* entry, u32 x, u32 y) = 0;
+ virtual void CopyRectangleFromTexture(
+ const TCacheEntryBase* source,
+ const MathUtil::Rectangle<int> &srcrect,
+ const MathUtil::Rectangle<int> &dstrect) = 0;
virtual void Load(unsigned int width, unsigned int height,
unsigned int expanded_width, unsigned int level) = 0;
- virtual void FromRenderTarget(u32 dstAddr, unsigned int dstFormat,
+ virtual void FromRenderTarget(u8* dst, unsigned int dstFormat, u32 dstStride,
PEControl::PixelFormat srcFormat, const EFBRectangle& srcRect,
bool isIntensity, bool scaleByHalf, unsigned int cbufid,
const float *colmat) = 0;
bool OverlapsMemoryRange(u32 range_address, u32 range_size) const;
- void DoPartialTextureUpdates();
-
bool IsEfbCopy() const { return is_efb_copy; }
+
+ u32 NumBlocksY() const;
+ u32 CacheLinesPerRow() const;
+
+ void Zero(u8* ptr);
+
+ u64 CalculateHash() const;
};
virtual ~TextureCache(); // needs virtual for DX11 dtor
@@ -114,7 +127,6 @@ public:
static void Cleanup(int _frameCount);
static void Invalidate();
- static void MakeRangeDynamic(u32 start_address, u32 size);
virtual TCacheEntryBase* CreateTexture(const TCacheEntryConfig& config) = 0;
@@ -124,8 +136,8 @@ public:
static TCacheEntryBase* Load(const u32 stage);
static void UnbindTextures();
static void BindTextures();
- static void CopyRenderTargetToTexture(u32 dstAddr, unsigned int dstFormat, PEControl::PixelFormat srcFormat,
- const EFBRectangle& srcRect, bool isIntensity, bool scaleByHalf);
+ static void CopyRenderTargetToTexture(u32 dstAddr, unsigned int dstFormat, u32 dstStride,
+ PEControl::PixelFormat srcFormat, const EFBRectangle& srcRect, bool isIntensity, bool scaleByHalf);
static void RequestInvalidateTextureCache();
@@ -134,13 +146,13 @@ public:
protected:
TextureCache();
- static GC_ALIGNED16(u8 *temp);
+ alignas(16) static u8* temp;
static size_t temp_size;
private:
typedef std::multimap<u64, TCacheEntryBase*> TexCache;
typedef std::unordered_multimap<TCacheEntryConfig, TCacheEntryBase*, TCacheEntryConfig::Hasher> TexPool;
-
+ static TCacheEntryBase* DoPartialTextureUpdates(TexCache::iterator iter);
static void DumpTexture(TCacheEntryBase* entry, std::string basename, unsigned int level);
static void CheckTempSize(size_t required_size);
diff --git a/Source/Core/VideoCommon/TextureConversionShader.cpp b/Source/Core/VideoCommon/TextureConversionShader.cpp
index f93d465a16..033a56b4aa 100644
--- a/Source/Core/VideoCommon/TextureConversionShader.cpp
+++ b/Source/Core/VideoCommon/TextureConversionShader.cpp
@@ -148,7 +148,7 @@ static void WriteToBitDepth(char*& p, u8 depth, const char* src, const char* des
WRITE(p, " %s = floor(%s * 255.0 / exp2(8.0 - %d.0));\n", dest, src, depth);
}
-static void WriteEncoderEnd(char*& p, API_TYPE ApiType)
+static void WriteEncoderEnd(char*& p)
{
WRITE(p, "}\n");
IntensityConstantAdded = false;
@@ -173,7 +173,7 @@ static void WriteI8Encoder(char*& p, API_TYPE ApiType)
WRITE(p, " ocol0.rgba += IntensityConst.aaaa;\n"); // see WriteColorToIntensity
- WriteEncoderEnd(p, ApiType);
+ WriteEncoderEnd(p);
}
static void WriteI4Encoder(char*& p, API_TYPE ApiType)
@@ -214,7 +214,7 @@ static void WriteI4Encoder(char*& p, API_TYPE ApiType)
WriteToBitDepth(p, 4, "color1", "color1");
WRITE(p, " ocol0 = (color0 * 16.0 + color1) / 255.0;\n");
- WriteEncoderEnd(p, ApiType);
+ WriteEncoderEnd(p);
}
static void WriteIA8Encoder(char*& p,API_TYPE ApiType)
@@ -232,7 +232,7 @@ static void WriteIA8Encoder(char*& p,API_TYPE ApiType)
WRITE(p, " ocol0.ga += IntensityConst.aa;\n");
- WriteEncoderEnd(p, ApiType);
+ WriteEncoderEnd(p);
}
static void WriteIA4Encoder(char*& p,API_TYPE ApiType)
@@ -264,7 +264,7 @@ static void WriteIA4Encoder(char*& p,API_TYPE ApiType)
WriteToBitDepth(p, 4, "color1", "color1");
WRITE(p, " ocol0 = (color0 * 16.0 + color1) / 255.0;\n");
- WriteEncoderEnd(p, ApiType);
+ WriteEncoderEnd(p);
}
static void WriteRGB565Encoder(char*& p,API_TYPE ApiType)
@@ -287,7 +287,7 @@ static void WriteRGB565Encoder(char*& p,API_TYPE ApiType)
WRITE(p, " ocol0.ga = ocol0.ga + gLower * 32.0;\n");
WRITE(p, " ocol0 = ocol0 / 255.0;\n");
- WriteEncoderEnd(p, ApiType);
+ WriteEncoderEnd(p);
}
static void WriteRGB5A3Encoder(char*& p,API_TYPE ApiType)
@@ -353,7 +353,7 @@ static void WriteRGB5A3Encoder(char*& p,API_TYPE ApiType)
WRITE(p, "}\n");
WRITE(p, " ocol0 = ocol0 / 255.0;\n");
- WriteEncoderEnd(p, ApiType);
+ WriteEncoderEnd(p);
}
static void WriteRGBA8Encoder(char*& p,API_TYPE ApiType)
@@ -378,7 +378,7 @@ static void WriteRGBA8Encoder(char*& p,API_TYPE ApiType)
WRITE(p, " ocol0 = first ? color0 : color1;\n");
- WriteEncoderEnd(p, ApiType);
+ WriteEncoderEnd(p);
}
static void WriteC4Encoder(char*& p, const char* comp,API_TYPE ApiType)
@@ -400,7 +400,7 @@ static void WriteC4Encoder(char*& p, const char* comp,API_TYPE ApiType)
WriteToBitDepth(p, 4, "color1", "color1");
WRITE(p, " ocol0 = (color0 * 16.0 + color1) / 255.0;\n");
- WriteEncoderEnd(p, ApiType);
+ WriteEncoderEnd(p);
}
static void WriteC8Encoder(char*& p, const char* comp,API_TYPE ApiType)
@@ -412,7 +412,7 @@ static void WriteC8Encoder(char*& p, const char* comp,API_TYPE ApiType)
WriteSampleColor(p, comp, "ocol0.r", 2, ApiType);
WriteSampleColor(p, comp, "ocol0.a", 3, ApiType);
- WriteEncoderEnd(p, ApiType);
+ WriteEncoderEnd(p);
}
static void WriteCC4Encoder(char*& p, const char* comp,API_TYPE ApiType)
@@ -442,7 +442,7 @@ static void WriteCC4Encoder(char*& p, const char* comp,API_TYPE ApiType)
WriteToBitDepth(p, 4, "color1", "color1");
WRITE(p, " ocol0 = (color0 * 16.0 + color1) / 255.0;\n");
- WriteEncoderEnd(p, ApiType);
+ WriteEncoderEnd(p);
}
static void WriteCC8Encoder(char*& p, const char* comp, API_TYPE ApiType)
@@ -452,7 +452,7 @@ static void WriteCC8Encoder(char*& p, const char* comp, API_TYPE ApiType)
WriteSampleColor(p, comp, "ocol0.bg", 0, ApiType);
WriteSampleColor(p, comp, "ocol0.ra", 1, ApiType);
- WriteEncoderEnd(p, ApiType);
+ WriteEncoderEnd(p);
}
static void WriteZ8Encoder(char*& p, const char* multiplier,API_TYPE ApiType)
@@ -477,7 +477,7 @@ static void WriteZ8Encoder(char*& p, const char* multiplier,API_TYPE ApiType)
if (ApiType == API_D3D) WRITE(p, "depth = 1.0f - depth;\n");
WRITE(p, "ocol0.a = frac(depth * %s);\n", multiplier);
- WriteEncoderEnd(p, ApiType);
+ WriteEncoderEnd(p);
}
static void WriteZ16Encoder(char*& p,API_TYPE ApiType)
@@ -492,7 +492,7 @@ static void WriteZ16Encoder(char*& p,API_TYPE ApiType)
WriteSampleColor(p, "r", "depth", 0, ApiType);
if (ApiType == API_D3D) WRITE(p, "depth = 1.0f - depth;\n");
- WRITE(p, " depth = clamp(depth * 16777216.0, 0.0, float(0xFFFFFF));\n");
+ WRITE(p, " depth *= 16777216.0;\n");
WRITE(p, " expanded.r = floor(depth / (256.0 * 256.0));\n");
WRITE(p, " depth -= expanded.r * 256.0 * 256.0;\n");
WRITE(p, " expanded.g = floor(depth / 256.0);\n");
@@ -503,7 +503,7 @@ static void WriteZ16Encoder(char*& p,API_TYPE ApiType)
WriteSampleColor(p, "r", "depth", 1, ApiType);
if (ApiType == API_D3D) WRITE(p, "depth = 1.0f - depth;\n");
- WRITE(p, " depth = clamp(depth * 16777216.0, 0.0, float(0xFFFFFF));\n");
+ WRITE(p, " depth *= 16777216.0;\n");
WRITE(p, " expanded.r = floor(depth / (256.0 * 256.0));\n");
WRITE(p, " depth -= expanded.r * 256.0 * 256.0;\n");
WRITE(p, " expanded.g = floor(depth / 256.0);\n");
@@ -511,7 +511,7 @@ static void WriteZ16Encoder(char*& p,API_TYPE ApiType)
WRITE(p, " ocol0.r = expanded.g / 255.0;\n");
WRITE(p, " ocol0.a = expanded.r / 255.0;\n");
- WriteEncoderEnd(p, ApiType);
+ WriteEncoderEnd(p);
}
static void WriteZ16LEncoder(char*& p,API_TYPE ApiType)
@@ -526,7 +526,7 @@ static void WriteZ16LEncoder(char*& p,API_TYPE ApiType)
WriteSampleColor(p, "r", "depth", 0, ApiType);
if (ApiType == API_D3D) WRITE(p, "depth = 1.0f - depth;\n");
- WRITE(p, " depth = clamp(depth * 16777216.0, 0.0, float(0xFFFFFF));\n");
+ WRITE(p, " depth *= 16777216.0;\n");
WRITE(p, " expanded.r = floor(depth / (256.0 * 256.0));\n");
WRITE(p, " depth -= expanded.r * 256.0 * 256.0;\n");
WRITE(p, " expanded.g = floor(depth / 256.0);\n");
@@ -539,7 +539,7 @@ static void WriteZ16LEncoder(char*& p,API_TYPE ApiType)
WriteSampleColor(p, "r", "depth", 1, ApiType);
if (ApiType == API_D3D) WRITE(p, "depth = 1.0f - depth;\n");
- WRITE(p, " depth = clamp(depth * 16777216.0, 0.0, float(0xFFFFFF));\n");
+ WRITE(p, " depth *= 16777216.0;\n");
WRITE(p, " expanded.r = floor(depth / (256.0 * 256.0));\n");
WRITE(p, " depth -= expanded.r * 256.0 * 256.0;\n");
WRITE(p, " expanded.g = floor(depth / 256.0);\n");
@@ -549,7 +549,7 @@ static void WriteZ16LEncoder(char*& p,API_TYPE ApiType)
WRITE(p, " ocol0.r = expanded.b / 255.0;\n");
WRITE(p, " ocol0.a = expanded.g / 255.0;\n");
- WriteEncoderEnd(p, ApiType);
+ WriteEncoderEnd(p);
}
static void WriteZ24Encoder(char*& p, API_TYPE ApiType)
@@ -568,7 +568,7 @@ static void WriteZ24Encoder(char*& p, API_TYPE ApiType)
for (int i = 0; i < 2; i++)
{
- WRITE(p, " depth%i = clamp(depth%i * 16777216.0, 0.0, float(0xFFFFFF));\n", i, i);
+ WRITE(p, " depth%i *= 16777216.0;\n", i);
WRITE(p, " expanded%i.r = floor(depth%i / (256.0 * 256.0));\n", i, i);
WRITE(p, " depth%i -= expanded%i.r * 256.0 * 256.0;\n", i, i);
@@ -591,7 +591,7 @@ static void WriteZ24Encoder(char*& p, API_TYPE ApiType)
WRITE(p, " ocol0.a = expanded1.r / 255.0;\n");
WRITE(p, " }\n");
- WriteEncoderEnd(p, ApiType);
+ WriteEncoderEnd(p);
}
const char *GenerateEncodingShader(u32 format,API_TYPE ApiType)
@@ -650,9 +650,11 @@ const char *GenerateEncodingShader(u32 format,API_TYPE ApiType)
case GX_CTF_GB8:
WriteCC8Encoder(p, "gb", ApiType);
break;
+ case GX_CTF_Z8H:
case GX_TF_Z8:
WriteC8Encoder(p, "r", ApiType);
break;
+ case GX_CTF_Z16R:
case GX_TF_Z16:
WriteZ16Encoder(p, ApiType);
break;
diff --git a/Source/Core/VideoCommon/TextureDecoder.h b/Source/Core/VideoCommon/TextureDecoder.h
index 1abcc5705c..249024fa3a 100644
--- a/Source/Core/VideoCommon/TextureDecoder.h
+++ b/Source/Core/VideoCommon/TextureDecoder.h
@@ -12,10 +12,11 @@ enum
TMEM_SIZE = 1024 * 1024,
TMEM_LINE_SIZE = 32,
};
-extern GC_ALIGNED16(u8 texMem[TMEM_SIZE]);
+alignas(16) extern u8 texMem[TMEM_SIZE];
enum TextureFormat
{
+ // These are the texture formats that can be read by the texture mapper.
GX_TF_I4 = 0x0,
GX_TF_I8 = 0x1,
GX_TF_IA4 = 0x2,
@@ -28,14 +29,21 @@ enum TextureFormat
GX_TF_C14X2 = 0xA,
GX_TF_CMPR = 0xE,
- _GX_TF_CTF = 0x20, // copy-texture-format only (simply means linear?)
- _GX_TF_ZTF = 0x10, // Z-texture-format
+ _GX_TF_ZTF = 0x10, // flag for Z texture formats (used internally by dolphin)
- // these formats are also valid when copying targets
+ // Depth texture formats (which directly map to the equivalent colour format above.)
+ GX_TF_Z8 = 0x1 | _GX_TF_ZTF,
+ GX_TF_Z16 = 0x3 | _GX_TF_ZTF,
+ GX_TF_Z24X8 = 0x6 | _GX_TF_ZTF,
+
+ _GX_TF_CTF = 0x20, // flag for copy-texture-format only (used internally by dolphin)
+
+ // These are extra formats that can be used when copying from efb,
+ // they use one of texel formats from above, but pack diffrent data into them.
GX_CTF_R4 = 0x0 | _GX_TF_CTF,
GX_CTF_RA4 = 0x2 | _GX_TF_CTF,
GX_CTF_RA8 = 0x3 | _GX_TF_CTF,
- GX_CTF_YUVA8 = 0x6 | _GX_TF_CTF,
+ GX_CTF_YUVA8 = 0x6 | _GX_TF_CTF, // YUV 4:4:4 - Dolphin doesn't implement this format as no commercial games use it
GX_CTF_A8 = 0x7 | _GX_TF_CTF,
GX_CTF_R8 = 0x8 | _GX_TF_CTF,
GX_CTF_G8 = 0x9 | _GX_TF_CTF,
@@ -43,13 +51,12 @@ enum TextureFormat
GX_CTF_RG8 = 0xB | _GX_TF_CTF,
GX_CTF_GB8 = 0xC | _GX_TF_CTF,
- GX_TF_Z8 = 0x1 | _GX_TF_ZTF,
- GX_TF_Z16 = 0x3 | _GX_TF_ZTF,
- GX_TF_Z24X8 = 0x6 | _GX_TF_ZTF,
-
+ // extra depth texture formats that can be used for efb copies.
GX_CTF_Z4 = 0x0 | _GX_TF_ZTF | _GX_TF_CTF,
+ GX_CTF_Z8H = 0x8 | _GX_TF_ZTF | _GX_TF_CTF, // This produces an identical result to to GX_TF_Z8
GX_CTF_Z8M = 0x9 | _GX_TF_ZTF | _GX_TF_CTF,
GX_CTF_Z8L = 0xA | _GX_TF_ZTF | _GX_TF_CTF,
+ GX_CTF_Z16R = 0xB | _GX_TF_ZTF | _GX_TF_CTF, // Reversed version of GX_TF_Z16
GX_CTF_Z16L = 0xC | _GX_TF_ZTF | _GX_TF_CTF,
};
diff --git a/Source/Core/VideoCommon/TextureDecoder_Common.cpp b/Source/Core/VideoCommon/TextureDecoder_Common.cpp
index 092b8ec1e0..5f8f314a0b 100644
--- a/Source/Core/VideoCommon/TextureDecoder_Common.cpp
+++ b/Source/Core/VideoCommon/TextureDecoder_Common.cpp
@@ -5,8 +5,10 @@
#include <algorithm>
#include <cmath>
-#include "Common/Common.h"
-
+#include "Common/CommonFuncs.h"
+#include "Common/CommonTypes.h"
+#include "Common/MsgHandler.h"
+#include "Common/Logging/Log.h"
#include "VideoCommon/LookUpTables.h"
#include "VideoCommon/sfont.inc"
#include "VideoCommon/TextureDecoder.h"
@@ -16,7 +18,7 @@ static bool TexFmt_Overlay_Center = false;
// TRAM
// STATE_TO_SAVE
-GC_ALIGNED16(u8 texMem[TMEM_SIZE]);
+alignas(16) u8 texMem[TMEM_SIZE];
int TexDecoder_GetTexelSizeInNibbles(int format)
{
@@ -35,7 +37,6 @@ int TexDecoder_GetTexelSizeInNibbles(int format)
case GX_CTF_R4: return 1;
case GX_CTF_RA4: return 2;
case GX_CTF_RA8: return 4;
- case GX_CTF_YUVA8: return 8;
case GX_CTF_A8: return 2;
case GX_CTF_R8: return 2;
case GX_CTF_G8: return 2;
@@ -48,10 +49,14 @@ int TexDecoder_GetTexelSizeInNibbles(int format)
case GX_TF_Z24X8: return 8;
case GX_CTF_Z4: return 1;
+ case GX_CTF_Z8H: return 2;
case GX_CTF_Z8M: return 2;
case GX_CTF_Z8L: return 2;
+ case GX_CTF_Z16R: return 4;
case GX_CTF_Z16L: return 4;
- default: return 1;
+ default:
+ PanicAlert("Unsupported Texture Format (%08x)! (GetTexelSizeInNibbles)", format);
+ return 1;
}
}
@@ -88,11 +93,13 @@ int TexDecoder_GetBlockWidthInTexels(u32 format)
case GX_TF_Z16: return 4;
case GX_TF_Z24X8: return 4;
case GX_CTF_Z4: return 8;
+ case GX_CTF_Z8H: return 8;
case GX_CTF_Z8M: return 8;
case GX_CTF_Z8L: return 8;
+ case GX_CTF_Z16R: return 4;
case GX_CTF_Z16L: return 4;
default:
- ERROR_LOG(VIDEO, "Unsupported Texture Format (%08x)! (GetBlockWidthInTexels)", format);
+ PanicAlert("Unsupported Texture Format (%08x)! (GetBlockWidthInTexels)", format);
return 8;
}
}
@@ -125,11 +132,13 @@ int TexDecoder_GetBlockHeightInTexels(u32 format)
case GX_TF_Z16: return 4;
case GX_TF_Z24X8: return 4;
case GX_CTF_Z4: return 8;
+ case GX_CTF_Z8H: return 4;
case GX_CTF_Z8M: return 4;
case GX_CTF_Z8L: return 4;
+ case GX_CTF_Z16R: return 4;
case GX_CTF_Z16L: return 4;
default:
- ERROR_LOG(VIDEO, "Unsupported Texture Format (%08x)! (GetBlockHeightInTexels)", format);
+ PanicAlert("Unsupported Texture Format (%08x)! (GetBlockHeightInTexels)", format);
return 4;
}
}
diff --git a/Source/Core/VideoCommon/TextureDecoder_x64.cpp b/Source/Core/VideoCommon/TextureDecoder_x64.cpp
index f68d35f842..3064b0f3c1 100644
--- a/Source/Core/VideoCommon/TextureDecoder_x64.cpp
+++ b/Source/Core/VideoCommon/TextureDecoder_x64.cpp
@@ -1065,8 +1065,8 @@ void _TexDecoder_DecodeImpl(u32 * dst, const u8 * src, int width, int height, in
const __m128i dxt = _mm_loadu_si128((__m128i *)(src + sizeof(struct DXTBlock) * 2 * xStep));
// Copy the 2-bit indices from each DXT block:
- GC_ALIGNED16( u32 dxttmp[4] );
- _mm_store_si128((__m128i *)dxttmp, dxt);
+ alignas(16) u32 dxttmp[4];
+ _mm_store_si128((__m128i*)dxttmp, dxt);
u32 dxt0sel = dxttmp[1];
u32 dxt1sel = dxttmp[3];
@@ -1204,10 +1204,10 @@ void _TexDecoder_DecodeImpl(u32 * dst, const u8 * src, int width, int height, in
u32 *dst32 = ( dst + (y + z*4) * width + x );
// Copy the colors here:
- GC_ALIGNED16( u32 colors0[4] );
- GC_ALIGNED16( u32 colors1[4] );
- _mm_store_si128((__m128i *)colors0, mmcolors0);
- _mm_store_si128((__m128i *)colors1, mmcolors1);
+ alignas(16) u32 colors0[4];
+ alignas(16) u32 colors1[4];
+ _mm_store_si128((__m128i*)colors0, mmcolors0);
+ _mm_store_si128((__m128i*)colors1, mmcolors1);
// Row 0:
dst32[(width * 0) + 0] = colors0[(dxt0sel >> ((0*8)+6)) & 3];
diff --git a/Source/Core/VideoCommon/VertexLoader.cpp b/Source/Core/VideoCommon/VertexLoader.cpp
index faf4599fdd..5673d1aa90 100644
--- a/Source/Core/VideoCommon/VertexLoader.cpp
+++ b/Source/Core/VideoCommon/VertexLoader.cpp
@@ -2,6 +2,7 @@
// Licensed under GPLv2+
// Refer to the license.txt file included.
+#include "Common/Common.h"
#include "Common/CommonTypes.h"
#include "Common/MemoryUtil.h"
diff --git a/Source/Core/VideoCommon/VertexLoaderARM64.cpp b/Source/Core/VideoCommon/VertexLoaderARM64.cpp
index cf321baa60..4251acaaa0 100644
--- a/Source/Core/VideoCommon/VertexLoaderARM64.cpp
+++ b/Source/Core/VideoCommon/VertexLoaderARM64.cpp
@@ -21,7 +21,7 @@ ARM64Reg stride_reg = X11;
ARM64Reg arraybase_reg = X10;
ARM64Reg scale_reg = X9;
-static const float GC_ALIGNED16(scale_factors[]) =
+alignas(16) static const float scale_factors[] =
{
1.0 / (1ULL << 0), 1.0 / (1ULL << 1), 1.0 / (1ULL << 2), 1.0 / (1ULL << 3),
1.0 / (1ULL << 4), 1.0 / (1ULL << 5), 1.0 / (1ULL << 6), 1.0 / (1ULL << 7),
@@ -47,17 +47,36 @@ VertexLoaderARM64::VertexLoaderARM64(const TVtxDesc& vtx_desc, const VAT& vtx_at
void VertexLoaderARM64::GetVertexAddr(int array, u64 attribute, ARM64Reg reg)
{
- ADD(reg, src_reg, m_src_ofs);
if (attribute & MASK_INDEXED)
{
if (attribute == INDEX8)
{
- LDRB(INDEX_UNSIGNED, scratch1_reg, reg, 0);
+ if (m_src_ofs < 4096)
+ {
+ LDRB(INDEX_UNSIGNED, scratch1_reg, src_reg, m_src_ofs);
+ }
+ else
+ {
+ ADD(reg, src_reg, m_src_ofs);
+ LDRB(INDEX_UNSIGNED, scratch1_reg, reg, 0);
+ }
m_src_ofs += 1;
}
else
{
- LDRH(INDEX_UNSIGNED, scratch1_reg, reg, 0);
+ if (m_src_ofs < 256)
+ {
+ LDURH(scratch1_reg, src_reg, m_src_ofs);
+ }
+ else if (m_src_ofs <= 8190 && !(m_src_ofs & 1))
+ {
+ LDRH(INDEX_UNSIGNED, scratch1_reg, src_reg, m_src_ofs);
+ }
+ else
+ {
+ ADD(reg, src_reg, m_src_ofs);
+ LDRH(INDEX_UNSIGNED, scratch1_reg, reg, 0);
+ }
m_src_ofs += 2;
REV16(scratch1_reg, scratch1_reg);
}
@@ -74,6 +93,8 @@ void VertexLoaderARM64::GetVertexAddr(int array, u64 attribute, ARM64Reg reg)
LDR(INDEX_UNSIGNED, EncodeRegTo64(scratch2_reg), arraybase_reg, array * 8);
ADD(EncodeRegTo64(reg), EncodeRegTo64(scratch1_reg), EncodeRegTo64(scratch2_reg));
}
+ else
+ ADD(reg, src_reg, m_src_ofs);
}
s32 VertexLoaderARM64::GetAddressImm(int array, u64 attribute, Arm64Gen::ARM64Reg reg, u32 align)
@@ -171,8 +192,7 @@ int VertexLoaderARM64::ReadVertex(u64 attribute, int format, int count_in, int c
CMP(count_reg, 3);
FixupBranch dont_store = B(CC_GT);
MOVI2R(EncodeRegTo64(scratch2_reg), (u64)VertexLoaderManager::position_cache);
- ORR(scratch1_reg, WSP, count_reg, ArithOption(count_reg, ST_LSL, 4));
- ADD(EncodeRegTo64(scratch1_reg), EncodeRegTo64(scratch1_reg), EncodeRegTo64(scratch2_reg));
+ ADD(EncodeRegTo64(scratch1_reg), EncodeRegTo64(scratch2_reg), EncodeRegTo64(count_reg), ArithOption(EncodeRegTo64(count_reg), ST_LSL, 4));
m_float_emit.STUR(write_size, coords, EncodeRegTo64(scratch1_reg), -16);
SetJumpTarget(dont_store);
}
@@ -347,12 +367,33 @@ void VertexLoaderARM64::GenerateVertexLoader()
// We can touch all except v8-v15
// If we need to use those, we need to retain the lower 64bits(!) of the register
- MOV(skipped_reg, WSP);
+ const u64 tc[8] = {
+ m_VtxDesc.Tex0Coord, m_VtxDesc.Tex1Coord, m_VtxDesc.Tex2Coord, m_VtxDesc.Tex3Coord,
+ m_VtxDesc.Tex4Coord, m_VtxDesc.Tex5Coord, m_VtxDesc.Tex6Coord, m_VtxDesc.Tex7Coord,
+ };
+
+ bool has_tc = false;
+ bool has_tc_scale = false;
+ for (int i = 0; i < 8; i++)
+ {
+ has_tc |= tc[i];
+ has_tc_scale |= !!m_VtxAttr.texCoord[i].Frac;
+ }
+
+ bool need_scale = (m_VtxAttr.ByteDequant && m_VtxAttr.PosFrac) ||
+ (has_tc && has_tc_scale) ||
+ m_VtxDesc.Normal;
+
+ AlignCode16();
+ if (m_VtxDesc.Position & MASK_INDEXED)
+ MOV(skipped_reg, WZR);
MOV(saved_count, count_reg);
MOVI2R(stride_reg, (u64)&g_main_cp_state.array_strides);
MOVI2R(arraybase_reg, (u64)&VertexLoaderManager::cached_arraybases);
- MOVI2R(scale_reg, (u64)&scale_factors);
+
+ if (need_scale)
+ MOVI2R(scale_reg, (u64)&scale_factors);
const u8* loop_start = GetCodePtr();
@@ -465,10 +506,7 @@ void VertexLoaderARM64::GenerateVertexLoader()
}
}
- const u64 tc[8] = {
- m_VtxDesc.Tex0Coord, m_VtxDesc.Tex1Coord, m_VtxDesc.Tex2Coord, m_VtxDesc.Tex3Coord,
- m_VtxDesc.Tex4Coord, m_VtxDesc.Tex5Coord, m_VtxDesc.Tex6Coord, m_VtxDesc.Tex7Coord,
- };
+
for (int i = 0; i < 8; i++)
{
diff --git a/Source/Core/VideoCommon/VertexLoaderBase.cpp b/Source/Core/VideoCommon/VertexLoaderBase.cpp
index d5831bc8e5..681f43b4fa 100644
--- a/Source/Core/VideoCommon/VertexLoaderBase.cpp
+++ b/Source/Core/VideoCommon/VertexLoaderBase.cpp
@@ -120,7 +120,7 @@ void VertexLoaderBase::AppendToString(std::string *dest) const
i, m_VtxAttr.texCoord[i].Elements, posMode[tex_mode[i]], posFormats[m_VtxAttr.texCoord[i].Format]));
}
}
- dest->append(StringFromFormat(" - %i v\n", m_numLoadedVertices));
+ dest->append(StringFromFormat(" - %i v", m_numLoadedVertices));
}
// a hacky implementation to compare two vertex loaders
@@ -131,13 +131,14 @@ public:
: VertexLoaderBase(vtx_desc, vtx_attr), a(_a), b(_b)
{
m_initialized = a && b && a->IsInitialized() && b->IsInitialized();
- bool can_test = a->m_VertexSize == b->m_VertexSize &&
- a->m_native_components == b->m_native_components &&
- a->m_native_vtx_decl.stride == b->m_native_vtx_decl.stride;
if (m_initialized)
{
- if (can_test)
+ m_initialized = a->m_VertexSize == b->m_VertexSize &&
+ a->m_native_components == b->m_native_components &&
+ a->m_native_vtx_decl.stride == b->m_native_vtx_decl.stride;
+
+ if (m_initialized)
{
m_VertexSize = a->m_VertexSize;
m_native_components = a->m_native_components;
@@ -152,8 +153,6 @@ public:
b->m_VertexSize, b->m_native_components, b->m_native_vtx_decl.stride);
}
}
-
- m_initialized &= can_test;
}
~VertexLoaderTester() override
{
diff --git a/Source/Core/VideoCommon/VertexLoaderManager.cpp b/Source/Core/VideoCommon/VertexLoaderManager.cpp
index 46a0eeaab0..0099064ca5 100644
--- a/Source/Core/VideoCommon/VertexLoaderManager.cpp
+++ b/Source/Core/VideoCommon/VertexLoaderManager.cpp
@@ -109,7 +109,8 @@ void AppendListToString(std::string *dest)
dest->reserve(dest->size() + total_size);
for (const entry& entry : entries)
{
- dest->append(entry.text);
+ *dest += entry.text;
+ *dest += '\n';
}
}
diff --git a/Source/Core/VideoCommon/VertexLoaderUtils.h b/Source/Core/VideoCommon/VertexLoaderUtils.h
index 143740ddf4..db37e26c43 100644
--- a/Source/Core/VideoCommon/VertexLoaderUtils.h
+++ b/Source/Core/VideoCommon/VertexLoaderUtils.h
@@ -4,6 +4,7 @@
#pragma once
+#include <cstring>
#include "Common/Common.h"
#include "VideoCommon/VertexManagerBase.h"
@@ -24,10 +25,11 @@ __forceinline void DataSkip()
}
template <typename T>
-__forceinline T DataPeek(int _uOffset, u8** bufp = &g_video_buffer_read_ptr)
+__forceinline T DataPeek(int _uOffset, u8* bufp = g_video_buffer_read_ptr)
{
- auto const result = Common::FromBigEndian(*reinterpret_cast<T*>(*bufp + _uOffset));
- return result;
+ T result;
+ std::memcpy(&result, &bufp[_uOffset], sizeof(T));
+ return Common::FromBigEndian(result);
}
// TODO: kill these
@@ -49,7 +51,7 @@ __forceinline u32 DataPeek32(int _uOffset)
template <typename T>
__forceinline T DataRead(u8** bufp = &g_video_buffer_read_ptr)
{
- auto const result = DataPeek<T>(0, bufp);
+ auto const result = DataPeek<T>(0, *bufp);
*bufp += sizeof(T);
return result;
}
@@ -77,9 +79,10 @@ __forceinline u32 DataReadU32()
__forceinline u32 DataReadU32Unswapped()
{
- u32 tmp = *(u32*)g_video_buffer_read_ptr;
- g_video_buffer_read_ptr += 4;
- return tmp;
+ u32 result;
+ std::memcpy(&result, g_video_buffer_read_ptr, sizeof(u32));
+ g_video_buffer_read_ptr += sizeof(u32);
+ return result;
}
__forceinline u8* DataGetPosition()
@@ -90,6 +93,6 @@ __forceinline u8* DataGetPosition()
template <typename T>
__forceinline void DataWrite(T data)
{
- *(T*)g_vertex_manager_write_ptr = data;
+ std::memcpy(g_vertex_manager_write_ptr, &data, sizeof(T));
g_vertex_manager_write_ptr += sizeof(T);
}
diff --git a/Source/Core/VideoCommon/VertexLoaderX64.cpp b/Source/Core/VideoCommon/VertexLoaderX64.cpp
index a298d7e1dd..d55ec17fef 100644
--- a/Source/Core/VideoCommon/VertexLoaderX64.cpp
+++ b/Source/Core/VideoCommon/VertexLoaderX64.cpp
@@ -38,7 +38,7 @@ VertexLoaderX64::VertexLoaderX64(const TVtxDesc& vtx_desc, const VAT& vtx_att) :
if (!IsInitialized())
return;
- AllocCodeSpace(4096);
+ AllocCodeSpace(4096, false);
ClearCodeSpace();
GenerateVertexLoader();
WriteProtect();
@@ -53,21 +53,12 @@ OpArg VertexLoaderX64::GetVertexAddr(int array, u64 attribute)
OpArg data = MDisp(src_reg, m_src_ofs);
if (attribute & MASK_INDEXED)
{
- if (attribute == INDEX8)
- {
- MOVZX(64, 8, scratch1, data);
- m_src_ofs += 1;
- }
- else
- {
- MOV(16, R(scratch1), data);
- m_src_ofs += 2;
- BSWAP(16, scratch1);
- MOVZX(64, 16, scratch1, R(scratch1));
- }
+ int bits = attribute == INDEX8 ? 8 : 16;
+ LoadAndSwap(bits, scratch1, data);
+ m_src_ofs += bits / 8;
if (array == ARRAY_POSITION)
{
- CMP(attribute == INDEX8 ? 8 : 16, R(scratch1), Imm8(-1));
+ CMP(bits, R(scratch1), Imm8(-1));
m_skip_vertex = J_CC(CC_E, true);
}
IMUL(32, scratch1, MPIC(&g_main_cp_state.array_strides[array]));
diff --git a/Source/Core/VideoCommon/VertexLoader_Color.cpp b/Source/Core/VideoCommon/VertexLoader_Color.cpp
index df9d814357..0fbc639a53 100644
--- a/Source/Core/VideoCommon/VertexLoader_Color.cpp
+++ b/Source/Core/VideoCommon/VertexLoader_Color.cpp
@@ -2,6 +2,10 @@
// Licensed under GPLv2+
// Refer to the license.txt file included.
+#include <cstring>
+
+#include "Common/Common.h"
+#include "Common/CommonFuncs.h"
#include "Common/CommonTypes.h"
#include "VideoCommon/VertexLoader.h"
@@ -12,15 +16,15 @@
#define AMASK 0xFF000000
-__forceinline void _SetCol(VertexLoader* loader, u32 val)
+static void SetCol(VertexLoader* loader, u32 val)
{
DataWrite(val);
loader->m_colIndex++;
}
-//color comes in format BARG in 16 bits
-//BARG -> AABBGGRR
-__forceinline void _SetCol4444(VertexLoader* loader, u16 val_)
+// Color comes in format BARG in 16 bits
+// BARG -> AABBGGRR
+static void SetCol4444(VertexLoader* loader, u16 val_)
{
u32 col, val = val_;
col = val & 0x00F0; // col = 000000R0;
@@ -28,24 +32,24 @@ __forceinline void _SetCol4444(VertexLoader* loader, u16 val_)
col |= (val & 0xF000) << 8; // col |= 00B00000;
col |= (val & 0x0F00) << 20; // col |= A0000000;
col |= col >> 4; // col = A0B0G0R0 | 0A0B0G0R;
- _SetCol(loader, col);
+ SetCol(loader, col);
}
-//color comes in format RGBA
-//RRRRRRGG GGGGBBBB BBAAAAAA
-__forceinline void _SetCol6666(VertexLoader* loader, u32 val)
+// Color comes in format RGBA
+// RRRRRRGG GGGGBBBB BBAAAAAA
+static void SetCol6666(VertexLoader* loader, u32 val)
{
u32 col = (val >> 16) & 0x000000FC;
col |= (val >> 2) & 0x0000FC00;
col |= (val << 12) & 0x00FC0000;
col |= (val << 26) & 0xFC000000;
col |= (col >> 6) & 0x03030303;
- _SetCol(loader, col);
+ SetCol(loader, col);
}
-//color comes in RGB
-//RRRRRGGG GGGBBBBB
-__forceinline void _SetCol565(VertexLoader* loader, u16 val_)
+// Color comes in RGB
+// RRRRRGGG GGGBBBBB
+static void SetCol565(VertexLoader* loader, u16 val_)
{
u32 col, val = val_;
col = (val >> 8) & 0x0000F8;
@@ -53,56 +57,64 @@ __forceinline void _SetCol565(VertexLoader* loader, u16 val_)
col |= (val << 19) & 0xF80000;
col |= (col >> 5) & 0x070007;
col |= (col >> 6) & 0x000300;
- _SetCol(loader, col | AMASK);
+ SetCol(loader, col | AMASK);
}
-__forceinline u32 _Read24(const u8 *addr)
+static u32 Read32(const u8* addr)
{
- return (*(const u32 *)addr) | AMASK;
+ u32 value;
+ std::memcpy(&value, addr, sizeof(u32));
+ return value;
}
-__forceinline u32 _Read32(const u8 *addr)
+static u32 Read24(const u8* addr)
{
- return *(const u32 *)addr;
+ return Read32(addr) | AMASK;
}
-
void Color_ReadDirect_24b_888(VertexLoader* loader)
{
- _SetCol(loader, _Read24(DataGetPosition()));
+ SetCol(loader, Read24(DataGetPosition()));
DataSkip(3);
}
void Color_ReadDirect_32b_888x(VertexLoader* loader)
{
- _SetCol(loader, _Read24(DataGetPosition()));
+ SetCol(loader, Read24(DataGetPosition()));
DataSkip(4);
}
void Color_ReadDirect_16b_565(VertexLoader* loader)
{
- _SetCol565(loader, DataReadU16());
+ SetCol565(loader, DataReadU16());
}
void Color_ReadDirect_16b_4444(VertexLoader* loader)
{
- _SetCol4444(loader, *(u16*)DataGetPosition());
+ u16 value;
+ std::memcpy(&value, DataGetPosition(), sizeof(u16));
+
+ SetCol4444(loader, value);
DataSkip(2);
}
void Color_ReadDirect_24b_6666(VertexLoader* loader)
{
- _SetCol6666(loader, Common::swap32(DataGetPosition() - 1));
+ SetCol6666(loader, Common::swap32(DataGetPosition() - 1));
DataSkip(3);
}
void Color_ReadDirect_32b_8888(VertexLoader* loader)
{
- _SetCol(loader, DataReadU32Unswapped());
+ SetCol(loader, DataReadU32Unswapped());
}
template <typename I>
void Color_ReadIndex_16b_565(VertexLoader* loader)
{
auto const Index = DataRead<I>();
- u16 val = Common::swap16(*(const u16 *)(VertexLoaderManager::cached_arraybases[ARRAY_COLOR + loader->m_colIndex] + (Index * g_main_cp_state.array_strides[ARRAY_COLOR + loader->m_colIndex])));
- _SetCol565(loader, val);
+ const u8* const address = VertexLoaderManager::cached_arraybases[ARRAY_COLOR + loader->m_colIndex] + (Index * g_main_cp_state.array_strides[ARRAY_COLOR + loader->m_colIndex]);
+
+ u16 value;
+ std::memcpy(&value, address, sizeof(u16));
+
+ SetCol565(loader, Common::swap16(value));
}
template <typename I>
@@ -110,7 +122,7 @@ void Color_ReadIndex_24b_888(VertexLoader* loader)
{
auto const Index = DataRead<I>();
const u8 *iAddress = VertexLoaderManager::cached_arraybases[ARRAY_COLOR + loader->m_colIndex] + (Index * g_main_cp_state.array_strides[ARRAY_COLOR + loader->m_colIndex]);
- _SetCol(loader, _Read24(iAddress));
+ SetCol(loader, Read24(iAddress));
}
template <typename I>
@@ -118,15 +130,19 @@ void Color_ReadIndex_32b_888x(VertexLoader* loader)
{
auto const Index = DataRead<I>();
const u8 *iAddress = VertexLoaderManager::cached_arraybases[ARRAY_COLOR + loader->m_colIndex] + (Index * g_main_cp_state.array_strides[ARRAY_COLOR + loader->m_colIndex]);
- _SetCol(loader, _Read24(iAddress));
+ SetCol(loader, Read24(iAddress));
}
template <typename I>
void Color_ReadIndex_16b_4444(VertexLoader* loader)
{
auto const Index = DataRead<I>();
- u16 val = *(const u16 *)(VertexLoaderManager::cached_arraybases[ARRAY_COLOR + loader->m_colIndex] + (Index * g_main_cp_state.array_strides[ARRAY_COLOR + loader->m_colIndex]));
- _SetCol4444(loader, val);
+ const u8* const address = VertexLoaderManager::cached_arraybases[ARRAY_COLOR + loader->m_colIndex] + (Index * g_main_cp_state.array_strides[ARRAY_COLOR + loader->m_colIndex]);
+
+ u16 value;
+ std::memcpy(&value, address, sizeof(u16));
+
+ SetCol4444(loader, value);
}
template <typename I>
@@ -135,7 +151,7 @@ void Color_ReadIndex_24b_6666(VertexLoader* loader)
auto const Index = DataRead<I>();
const u8* pData = VertexLoaderManager::cached_arraybases[ARRAY_COLOR + loader->m_colIndex] + (Index * g_main_cp_state.array_strides[ARRAY_COLOR + loader->m_colIndex]) - 1;
u32 val = Common::swap32(pData);
- _SetCol6666(loader, val);
+ SetCol6666(loader, val);
}
template <typename I>
@@ -143,7 +159,7 @@ void Color_ReadIndex_32b_8888(VertexLoader* loader)
{
auto const Index = DataRead<I>();
const u8 *iAddress = VertexLoaderManager::cached_arraybases[ARRAY_COLOR + loader->m_colIndex] + (Index * g_main_cp_state.array_strides[ARRAY_COLOR + loader->m_colIndex]);
- _SetCol(loader, _Read32(iAddress));
+ SetCol(loader, Read32(iAddress));
}
void Color_ReadIndex8_16b_565(VertexLoader* loader) { Color_ReadIndex_16b_565<u8>(loader); }
diff --git a/Source/Core/VideoCommon/VertexLoader_Normal.cpp b/Source/Core/VideoCommon/VertexLoader_Normal.cpp
index 615a09c3a9..5561d51c7a 100644
--- a/Source/Core/VideoCommon/VertexLoader_Normal.cpp
+++ b/Source/Core/VideoCommon/VertexLoader_Normal.cpp
@@ -5,6 +5,7 @@
#include <cmath>
#include <type_traits>
+#include "Common/Common.h"
#include "Common/CommonTypes.h"
#include "VideoCommon/VertexLoader.h"
#include "VideoCommon/VertexLoader_Normal.h"
diff --git a/Source/Core/VideoCommon/VertexLoader_Position.cpp b/Source/Core/VideoCommon/VertexLoader_Position.cpp
index 906cf61d71..609038ab3e 100644
--- a/Source/Core/VideoCommon/VertexLoader_Position.cpp
+++ b/Source/Core/VideoCommon/VertexLoader_Position.cpp
@@ -4,6 +4,7 @@
#include <type_traits>
+#include "Common/Common.h"
#include "Common/CommonTypes.h"
#include "VideoCommon/VertexLoader.h"
#include "VideoCommon/VertexLoader_Position.h"
diff --git a/Source/Core/VideoCommon/VertexLoader_Position.h b/Source/Core/VideoCommon/VertexLoader_Position.h
index 65a97e5bde..306a1a8c4a 100644
--- a/Source/Core/VideoCommon/VertexLoader_Position.h
+++ b/Source/Core/VideoCommon/VertexLoader_Position.h
@@ -6,12 +6,9 @@
#include "VideoCommon/NativeVertexFormat.h"
-class VertexLoader_Position {
+class VertexLoader_Position
+{
public:
-
- // Init
- static void Init();
-
// GetSize
static unsigned int GetSize(u64 _type, unsigned int _format, unsigned int _elements);
diff --git a/Source/Core/VideoCommon/VertexLoader_TextCoord.cpp b/Source/Core/VideoCommon/VertexLoader_TextCoord.cpp
index 28a9a33871..764efe7279 100644
--- a/Source/Core/VideoCommon/VertexLoader_TextCoord.cpp
+++ b/Source/Core/VideoCommon/VertexLoader_TextCoord.cpp
@@ -4,6 +4,7 @@
#include <type_traits>
+#include "Common/Common.h"
#include "Common/CommonTypes.h"
#include "VideoCommon/VertexLoader.h"
#include "VideoCommon/VertexLoader_TextCoord.h"
diff --git a/Source/Core/VideoCommon/VertexLoader_TextCoord.h b/Source/Core/VideoCommon/VertexLoader_TextCoord.h
index 0e3c7bac62..4d8d352058 100644
--- a/Source/Core/VideoCommon/VertexLoader_TextCoord.h
+++ b/Source/Core/VideoCommon/VertexLoader_TextCoord.h
@@ -9,10 +9,6 @@
class VertexLoader_TextCoord
{
public:
-
- // Init
- static void Init();
-
// GetSize
static unsigned int GetSize(u64 _type, unsigned int _format, unsigned int _elements);
diff --git a/Source/Core/VideoCommon/VertexManagerBase.cpp b/Source/Core/VideoCommon/VertexManagerBase.cpp
index e9e8ebf901..6358ed371c 100644
--- a/Source/Core/VideoCommon/VertexManagerBase.cpp
+++ b/Source/Core/VideoCommon/VertexManagerBase.cpp
@@ -250,8 +250,7 @@ void VertexManager::Flush()
GeometryShaderManager::SetConstants();
PixelShaderManager::SetConstants();
- bool useDstAlpha = !g_ActiveConfig.bDstAlphaPass &&
- bpmem.dstalpha.enable &&
+ bool useDstAlpha = bpmem.dstalpha.enable &&
bpmem.blendmode.alphaupdate &&
bpmem.zcontrol.pixel_format == PEControl::RGBA6_Z24;
diff --git a/Source/Core/VideoCommon/VertexShaderGen.cpp b/Source/Core/VideoCommon/VertexShaderGen.cpp
index 97ac15e887..63f80488e4 100644
--- a/Source/Core/VideoCommon/VertexShaderGen.cpp
+++ b/Source/Core/VideoCommon/VertexShaderGen.cpp
@@ -32,29 +32,6 @@ static inline void GenerateVertexShader(T& out, u32 components, API_TYPE api_typ
_assert_(bpmem.genMode.numtexgens == xfmem.numTexGen.numTexGens);
_assert_(bpmem.genMode.numcolchans == xfmem.numChan.numColorChans);
- if (DriverDetails::HasBug(DriverDetails::BUG_BROKENIVECSHIFTS))
- {
- // Add functions to do shifts on scalars and ivecs.
- // This is included in the vertex shader for lighting shader generation.
- out.Write("int ilshift(int a, int b) { return a << b; }\n"
- "int irshift(int a, int b) { return a >> b; }\n"
-
- "int2 ilshift(int2 a, int2 b) { return int2(a.x << b.x, a.y << b.y); }\n"
- "int2 ilshift(int2 a, int b) { return int2(a.x << b, a.y << b); }\n"
- "int2 irshift(int2 a, int2 b) { return int2(a.x >> b.x, a.y >> b.y); }\n"
- "int2 irshift(int2 a, int b) { return int2(a.x >> b, a.y >> b); }\n"
-
- "int3 ilshift(int3 a, int3 b) { return int3(a.x << b.x, a.y << b.y, a.z << b.z); }\n"
- "int3 ilshift(int3 a, int b) { return int3(a.x << b, a.y << b, a.z << b); }\n"
- "int3 irshift(int3 a, int3 b) { return int3(a.x >> b.x, a.y >> b.y, a.z >> b.z); }\n"
- "int3 irshift(int3 a, int b) { return int3(a.x >> b, a.y >> b, a.z >> b); }\n"
-
- "int4 ilshift(int4 a, int4 b) { return int4(a.x << b.x, a.y << b.y, a.z << b.z, a.w << b.w); }\n"
- "int4 ilshift(int4 a, int b) { return int4(a.x << b, a.y << b, a.z << b, a.w << b); }\n"
- "int4 irshift(int4 a, int4 b) { return int4(a.x >> b.x, a.y >> b.y, a.z >> b.z, a.w >> b.w); }\n"
- "int4 irshift(int4 a, int b) { return int4(a.x >> b, a.y >> b, a.z >> b, a.w >> b); }\n\n");
- }
-
out.Write("%s", s_lighting_struct);
// uniforms
@@ -100,7 +77,7 @@ static inline void GenerateVertexShader(T& out, u32 components, API_TYPE api_typ
if (g_ActiveConfig.backend_info.bSupportsGeometryShaders)
{
out.Write("out VertexData {\n");
- GenerateVSOutputMembers<T>(out, api_type, g_ActiveConfig.backend_info.bSupportsBindingLayout ? "centroid" : "centroid out");
+ GenerateVSOutputMembers<T>(out, api_type, GetInterpolationQualifier(api_type, false, true));
out.Write("} vs;\n");
}
else
@@ -110,17 +87,17 @@ static inline void GenerateVertexShader(T& out, u32 components, API_TYPE api_typ
{
if (i < xfmem.numTexGen.numTexGens)
{
- out.Write("centroid out float3 uv%d;\n", i);
+ out.Write("%s out float3 uv%d;\n", GetInterpolationQualifier(api_type), i);
}
}
- out.Write("centroid out float4 clipPos;\n");
+ out.Write("%s out float4 clipPos;\n", GetInterpolationQualifier(api_type));
if (g_ActiveConfig.bEnablePixelLighting)
{
- out.Write("centroid out float3 Normal;\n");
- out.Write("centroid out float3 WorldPos;\n");
+ out.Write("%s out float3 Normal;\n", GetInterpolationQualifier(api_type));
+ out.Write("%s out float3 WorldPos;\n", GetInterpolationQualifier(api_type));
}
- out.Write("centroid out float4 colors_0;\n");
- out.Write("centroid out float4 colors_1;\n");
+ out.Write("%s out float4 colors_0;\n", GetInterpolationQualifier(api_type));
+ out.Write("%s out float4 colors_1;\n", GetInterpolationQualifier(api_type));
}
out.Write("void main()\n{\n");
@@ -156,22 +133,12 @@ static inline void GenerateVertexShader(T& out, u32 components, API_TYPE api_typ
// transforms
if (components & VB_HAS_POSMTXIDX)
{
- if (is_writing_shadercode && (DriverDetails::HasBug(DriverDetails::BUG_NODYNUBOACCESS) && !DriverDetails::HasBug(DriverDetails::BUG_ANNIHILATEDUBOS)))
- {
- // This'll cause issues, but it can't be helped
- out.Write("float4 pos = float4(dot(" I_TRANSFORMMATRICES"[0], rawpos), dot(" I_TRANSFORMMATRICES"[1], rawpos), dot(" I_TRANSFORMMATRICES"[2], rawpos), 1);\n");
- if (components & VB_HAS_NRMALL)
- out.Write("float3 N0 = " I_NORMALMATRICES"[0].xyz, N1 = " I_NORMALMATRICES"[1].xyz, N2 = " I_NORMALMATRICES"[2].xyz;\n");
- }
- else
- {
- out.Write("float4 pos = float4(dot(" I_TRANSFORMMATRICES"[posmtx], rawpos), dot(" I_TRANSFORMMATRICES"[posmtx+1], rawpos), dot(" I_TRANSFORMMATRICES"[posmtx+2], rawpos), 1);\n");
+ out.Write("float4 pos = float4(dot(" I_TRANSFORMMATRICES"[posmtx], rawpos), dot(" I_TRANSFORMMATRICES"[posmtx+1], rawpos), dot(" I_TRANSFORMMATRICES"[posmtx+2], rawpos), 1);\n");
- if (components & VB_HAS_NRMALL)
- {
- out.Write("int normidx = posmtx >= 32 ? (posmtx-32) : posmtx;\n");
- out.Write("float3 N0 = " I_NORMALMATRICES"[normidx].xyz, N1 = " I_NORMALMATRICES"[normidx+1].xyz, N2 = " I_NORMALMATRICES"[normidx+2].xyz;\n");
- }
+ if (components & VB_HAS_NRMALL)
+ {
+ out.Write("int normidx = posmtx >= 32 ? (posmtx-32) : posmtx;\n");
+ out.Write("float3 N0 = " I_NORMALMATRICES"[normidx].xyz, N1 = " I_NORMALMATRICES"[normidx+1].xyz, N2 = " I_NORMALMATRICES"[normidx+2].xyz;\n");
}
if (components & VB_HAS_NRM0)
@@ -241,7 +208,8 @@ static inline void GenerateVertexShader(T& out, u32 components, API_TYPE api_typ
switch (texinfo.sourcerow)
{
case XF_SRCGEOM_INROW:
- _assert_(texinfo.inputform == XF_TEXINPUT_ABC1);
+ // The following assert was triggered in Super Smash Bros. Project M 3.6.
+ //_assert_(texinfo.inputform == XF_TEXINPUT_ABC1);
out.Write("coord = rawpos;\n"); // pos.w is 1
break;
case XF_SRCNORMAL_INROW:
@@ -291,7 +259,8 @@ static inline void GenerateVertexShader(T& out, u32 components, API_TYPE api_typ
}
else
{
- _assert_(0); // should have normals
+ // The following assert was triggered in House of the Dead Overkill and Star Wars Rogue Squadron 2
+ //_assert_(0); // should have normals
uid_data->texMtxInfo[i].embosssourceshift = xfmem.texMtxInfo[i].embosssourceshift;
out.Write("o.tex%d.xyz = o.tex%d.xyz;\n", i, texinfo.embosssourceshift);
}
diff --git a/Source/Core/VideoCommon/VertexShaderManager.cpp b/Source/Core/VideoCommon/VertexShaderManager.cpp
index 509974c373..93d248938b 100644
--- a/Source/Core/VideoCommon/VertexShaderManager.cpp
+++ b/Source/Core/VideoCommon/VertexShaderManager.cpp
@@ -9,6 +9,8 @@
#include "Common/BitSet.h"
#include "Common/CommonTypes.h"
#include "Common/MathUtil.h"
+#include "Core/ConfigManager.h"
+#include "Core/Core.h"
#include "VideoCommon/BPMemory.h"
#include "VideoCommon/CPMemory.h"
#include "VideoCommon/RenderBase.h"
@@ -20,7 +22,7 @@
#include "VideoCommon/VideoConfig.h"
#include "VideoCommon/XFMemory.h"
-static float GC_ALIGNED16(g_fProjectionMatrix[16]);
+alignas(16) static float g_fProjectionMatrix[16];
// track changes
static bool bTexMatricesChanged[2], bPosNormalMatrixChanged, bProjectionChanged, bViewportChanged;
@@ -87,6 +89,23 @@ static float PHackValue(std::string sValue)
return f;
}
+// Due to the BT.601 standard which the GameCube is based on being a compromise
+// between PAL and NTSC, neither standard gets square pixels. They are each off
+// by ~9% in opposite directions.
+// Just in case any game decides to take this into account, we do both these
+// tests with a large amount of slop.
+static bool AspectIs4_3(float width, float height)
+{
+ float aspect = fabsf(width / height);
+ return fabsf(aspect - 4.0f / 3.0f) < 4.0f / 3.0f * 0.11; // within 11% of 4:3
+}
+
+static bool AspectIs16_9(float width, float height)
+{
+ float aspect = fabsf(width / height);
+ return fabsf(aspect - 16.0f / 9.0f) < 16.0f / 9.0f * 0.11; // within 11% of 16:9
+}
+
void UpdateProjectionHack(int iPhackvalue[], std::string sPhackvalue[])
{
float fhackvalue1 = 0, fhackvalue2 = 0;
@@ -417,6 +436,16 @@ void VertexShaderManager::SetConstants()
g_fProjectionMatrix[14] = -1.0f;
g_fProjectionMatrix[15] = 0.0f;
+ // Heuristic to detect if a GameCube game is in 16:9 anamorphic widescreen mode.
+ if (!SConfig::GetInstance().bWii)
+ {
+ bool viewport_is_4_3 = AspectIs4_3(xfmem.viewport.wd, xfmem.viewport.ht);
+ if (AspectIs16_9(rawProjection[2], rawProjection[0]) && viewport_is_4_3)
+ g_aspect_wide = true; // Projection is 16:9 and viewport is 4:3, we are rendering an anamorphic widescreen picture
+ else if (AspectIs4_3(rawProjection[2], rawProjection[0]) && viewport_is_4_3)
+ g_aspect_wide = false; // Project and viewports are both 4:3, we are rendering a normal image.
+ }
+
SETSTAT_FT(stats.gproj_0, g_fProjectionMatrix[0]);
SETSTAT_FT(stats.gproj_1, g_fProjectionMatrix[1]);
SETSTAT_FT(stats.gproj_2, g_fProjectionMatrix[2]);
@@ -654,7 +683,7 @@ void VertexShaderManager::SetProjectionChanged()
bProjectionChanged = true;
}
-void VertexShaderManager::SetMaterialColorChanged(int index, u32 color)
+void VertexShaderManager::SetMaterialColorChanged(int index)
{
nMaterialsChanged[index] = true;
}
diff --git a/Source/Core/VideoCommon/VertexShaderManager.h b/Source/Core/VideoCommon/VertexShaderManager.h
index 325b18a2cd..35c146c7aa 100644
--- a/Source/Core/VideoCommon/VertexShaderManager.h
+++ b/Source/Core/VideoCommon/VertexShaderManager.h
@@ -28,7 +28,7 @@ public:
static void SetTexMatrixChangedB(u32 value);
static void SetViewportChanged();
static void SetProjectionChanged();
- static void SetMaterialColorChanged(int index, u32 color);
+ static void SetMaterialColorChanged(int index);
static void TranslateView(float x, float y, float z = 0.0f);
static void RotateView(float x, float y);
diff --git a/Source/Core/VideoCommon/VideoBackendBase.cpp b/Source/Core/VideoCommon/VideoBackendBase.cpp
index c0e466cb5c..69beee3f1d 100644
--- a/Source/Core/VideoCommon/VideoBackendBase.cpp
+++ b/Source/Core/VideoCommon/VideoBackendBase.cpp
@@ -18,21 +18,6 @@ static VideoBackend* s_default_backend = nullptr;
#ifdef _WIN32
#include <windows.h>
-// http://msdn.microsoft.com/en-us/library/ms725491.aspx
-static bool IsGteVista()
-{
- OSVERSIONINFOEX osvi;
- DWORDLONG dwlConditionMask = 0;
-
- ZeroMemory(&osvi, sizeof(OSVERSIONINFOEX));
- osvi.dwOSVersionInfoSize = sizeof(OSVERSIONINFOEX);
- osvi.dwMajorVersion = 6;
-
- VER_SET_CONDITION(dwlConditionMask, VER_MAJORVERSION, VER_GREATER_EQUAL);
-
- return VerifyVersionInfo(&osvi, VER_MAJORVERSION, dwlConditionMask) != FALSE;
-}
-
// Nvidia drivers >= v302 will check if the application exports a global
// variable named NvOptimusEnablement to know if it should run the app in high
// performance graphics mode or using the IGP.
@@ -48,8 +33,7 @@ void VideoBackend::PopulateList()
// OGL > D3D11 > SW
g_available_video_backends.push_back(backends[0] = new OGL::VideoBackend);
#ifdef _WIN32
- if (IsGteVista())
- g_available_video_backends.push_back(backends[1] = new DX11::VideoBackend);
+ g_available_video_backends.push_back(backends[1] = new DX11::VideoBackend);
#endif
g_available_video_backends.push_back(backends[3] = new SW::VideoSoftware);
diff --git a/Source/Core/VideoCommon/VideoBackendBase.h b/Source/Core/VideoCommon/VideoBackendBase.h
index 4671afdc6b..f6cd1d12bf 100644
--- a/Source/Core/VideoCommon/VideoBackendBase.h
+++ b/Source/Core/VideoCommon/VideoBackendBase.h
@@ -15,9 +15,8 @@ namespace MMIO { class Mapping; }
enum FieldType
{
- FIELD_PROGRESSIVE = 0,
- FIELD_UPPER,
- FIELD_LOWER
+ FIELD_ODD = 0,
+ FIELD_EVEN = 1,
};
enum EFBAccessType
@@ -161,5 +160,4 @@ public:
protected:
void InitializeShared();
- void InvalidState();
};
diff --git a/Source/Core/VideoCommon/VideoCommon.h b/Source/Core/VideoCommon/VideoCommon.h
index b81ebb8929..ba339e5c4f 100644
--- a/Source/Core/VideoCommon/VideoCommon.h
+++ b/Source/Core/VideoCommon/VideoCommon.h
@@ -12,6 +12,9 @@
#include "Common/MathUtil.h"
#include "VideoCommon/VideoBackendBase.h"
+// Global flag to signal if FifoRecorder is active.
+extern bool g_bRecordFifoData;
+
// These are accurate (disregarding AA modes).
enum
{
@@ -19,9 +22,10 @@ enum
EFB_HEIGHT = 528,
};
-// XFB width is decided by EFB copy operation. The VI can do horizontal
-// scaling (TODO: emulate).
-const u32 MAX_XFB_WIDTH = EFB_WIDTH;
+// Max XFB width is 720. You can only copy out 640 wide areas of efb to XFB
+// so you need multiple copies to do the full width.
+// The VI can do horizontal scaling (TODO: emulate).
+const u32 MAX_XFB_WIDTH = 720;
// Although EFB height is 528, 574-line XFB's can be created either with
// vertical scaling by the EFB copy operation or copying to multiple XFB's
@@ -65,12 +69,12 @@ struct TargetRectangle : public MathUtil::Rectangle<int>
#define LOG_VTX()
-typedef enum
+enum API_TYPE
{
API_OPENGL = 1,
API_D3D = 2,
API_NONE = 3
-} API_TYPE;
+};
inline u32 RGBA8ToRGBA6ToRGBA8(u32 src)
{
diff --git a/Source/Core/VideoCommon/VideoCommon.vcxproj b/Source/Core/VideoCommon/VideoCommon.vcxproj
index ff56a97703..98eecf8542 100644
--- a/Source/Core/VideoCommon/VideoCommon.vcxproj
+++ b/Source/Core/VideoCommon/VideoCommon.vcxproj
@@ -1,5 +1,5 @@
<?xml version="1.0" encoding="utf-8"?>
-<Project DefaultTargets="Build" ToolsVersion="12.0" xmlns="http://schemas.microsoft.com/developer/msbuild/2003">
+<Project DefaultTargets="Build" ToolsVersion="14.0" xmlns="http://schemas.microsoft.com/developer/msbuild/2003">
<ItemGroup Label="ProjectConfigurations">
<ProjectConfiguration Include="Debug|x64">
<Configuration>Debug</Configuration>
@@ -16,7 +16,7 @@
<Import Project="$(VCTargetsPath)\Microsoft.Cpp.Default.props" />
<PropertyGroup Label="Configuration">
<ConfigurationType>StaticLibrary</ConfigurationType>
- <PlatformToolset>v120</PlatformToolset>
+ <PlatformToolset>v140</PlatformToolset>
<CharacterSet>Unicode</CharacterSet>
</PropertyGroup>
<PropertyGroup Condition="'$(Configuration)'=='Debug'" Label="Configuration">
@@ -162,4 +162,4 @@
<Import Project="$(VCTargetsPath)\Microsoft.Cpp.targets" />
<ImportGroup Label="ExtensionTargets">
</ImportGroup>
-</Project>
+</Project> \ No newline at end of file
diff --git a/Source/Core/VideoCommon/VideoConfig.cpp b/Source/Core/VideoCommon/VideoConfig.cpp
index 92abca4819..6aadfbf075 100644
--- a/Source/Core/VideoCommon/VideoConfig.cpp
+++ b/Source/Core/VideoCommon/VideoConfig.cpp
@@ -78,8 +78,8 @@ void VideoConfig::Load(const std::string& ini_file)
settings->Get("EnablePixelLighting", &bEnablePixelLighting, 0);
settings->Get("FastDepthCalc", &bFastDepthCalc, true);
settings->Get("MSAA", &iMultisampleMode, 0);
+ settings->Get("SSAA", &bSSAA, false);
settings->Get("EFBScale", &iEFBScale, (int)SCALE_1X); // native
- settings->Get("DstAlphaPass", &bDstAlphaPass, false);
settings->Get("TexFmtOverlayEnable", &bTexFmtOverlayEnable, 0);
settings->Get("TexFmtOverlayCenter", &bTexFmtOverlayCenter, 0);
settings->Get("WireFrame", &bWireFrame, 0);
@@ -99,6 +99,7 @@ void VideoConfig::Load(const std::string& ini_file)
IniFile::Section* hacks = iniFile.GetOrCreateSection("Hacks");
hacks->Get("EFBAccessEnable", &bEFBAccessEnable, true);
hacks->Get("BBoxEnable", &bBBoxEnable, false);
+ hacks->Get("ForceProgressive", &bForceProgressive, true);
hacks->Get("EFBToTextureEnable", &bSkipEFBCopyToRam, true);
hacks->Get("EFBScaledCopy", &bCopyEFBScaled, true);
hacks->Get("EFBEmulateFormatChanges", &bEFBEmulateFormatChanges, false);
@@ -160,6 +161,8 @@ void VideoConfig::GameIniLoad()
CHECK_SETTING("Video_Settings", "EnablePixelLighting", bEnablePixelLighting);
CHECK_SETTING("Video_Settings", "FastDepthCalc", bFastDepthCalc);
CHECK_SETTING("Video_Settings", "MSAA", iMultisampleMode);
+ CHECK_SETTING("Video_Settings", "SSAA", bSSAA);
+
int tmp = -9000;
CHECK_SETTING("Video_Settings", "EFBScale", tmp); // integral
if (tmp != -9000)
@@ -187,7 +190,6 @@ void VideoConfig::GameIniLoad()
}
}
- CHECK_SETTING("Video_Settings", "DstAlphaPass", bDstAlphaPass);
CHECK_SETTING("Video_Settings", "DisableFog", bDisableFog);
CHECK_SETTING("Video_Enhancements", "ForceFiltering", bForceFiltering);
@@ -204,6 +206,7 @@ void VideoConfig::GameIniLoad()
CHECK_SETTING("Video_Hacks", "EFBAccessEnable", bEFBAccessEnable);
CHECK_SETTING("Video_Hacks", "BBoxEnable", bBBoxEnable);
+ CHECK_SETTING("Video_Hacks", "ForceProgressive", bForceProgressive);
CHECK_SETTING("Video_Hacks", "EFBToTextureEnable", bSkipEFBCopyToRam);
CHECK_SETTING("Video_Hacks", "EFBScaledCopy", bCopyEFBScaled);
CHECK_SETTING("Video_Hacks", "EFBEmulateFormatChanges", bEFBEmulateFormatChanges);
@@ -272,11 +275,11 @@ void VideoConfig::Save(const std::string& ini_file)
settings->Set("FastDepthCalc", bFastDepthCalc);
settings->Set("ShowEFBCopyRegions", bShowEFBCopyRegions);
settings->Set("MSAA", iMultisampleMode);
+ settings->Set("SSAA", bSSAA);
settings->Set("EFBScale", iEFBScale);
settings->Set("TexFmtOverlayEnable", bTexFmtOverlayEnable);
settings->Set("TexFmtOverlayCenter", bTexFmtOverlayCenter);
settings->Set("Wireframe", bWireFrame);
- settings->Set("DstAlphaPass", bDstAlphaPass);
settings->Set("DisableFog", bDisableFog);
settings->Set("EnableShaderDebugging", bEnableShaderDebugging);
settings->Set("BorderlessFullscreen", bBorderlessFullscreen);
@@ -293,6 +296,7 @@ void VideoConfig::Save(const std::string& ini_file)
IniFile::Section* hacks = iniFile.GetOrCreateSection("Hacks");
hacks->Set("EFBAccessEnable", bEFBAccessEnable);
hacks->Set("BBoxEnable", bBBoxEnable);
+ hacks->Set("ForceProgressive", bForceProgressive);
hacks->Set("EFBToTextureEnable", bSkipEFBCopyToRam);
hacks->Set("EFBScaledCopy", bCopyEFBScaled);
hacks->Set("EFBEmulateFormatChanges", bEFBEmulateFormatChanges);
diff --git a/Source/Core/VideoCommon/VideoConfig.h b/Source/Core/VideoCommon/VideoConfig.h
index 13451bbf5a..f6b4f9e7f0 100644
--- a/Source/Core/VideoCommon/VideoConfig.h
+++ b/Source/Core/VideoCommon/VideoConfig.h
@@ -25,10 +25,10 @@
enum AspectMode
{
- ASPECT_AUTO = 0,
- ASPECT_FORCE_16_9 = 1,
- ASPECT_FORCE_4_3 = 2,
- ASPECT_STRETCH = 3,
+ ASPECT_AUTO = 0,
+ ASPECT_ANALOG_WIDE = 1,
+ ASPECT_ANALOG = 2,
+ ASPECT_STRETCH = 3,
};
enum EFBScale
@@ -75,6 +75,7 @@ struct VideoConfig final
// Enhancements
int iMultisampleMode;
+ bool bSSAA;
int iEFBScale;
bool bForceFiltering;
int iMaxAnisotropy;
@@ -95,7 +96,6 @@ struct VideoConfig final
// Render
bool bWireFrame;
- bool bDstAlphaPass;
bool bDisableFog;
// Utility
@@ -112,6 +112,7 @@ struct VideoConfig final
bool bEFBAccessEnable;
bool bPerfQueriesEnable;
bool bBBoxEnable;
+ bool bForceProgressive;
bool bEFBEmulateFormatChanges;
bool bSkipEFBCopyToRam;
@@ -143,7 +144,7 @@ struct VideoConfig final
API_TYPE APIType;
std::vector<std::string> Adapters; // for D3D
- std::vector<std::string> AAModes;
+ std::vector<int> AAModes;
std::vector<std::string> PPShaders; // post-processing shaders
std::vector<std::string> AnaglyphShaders; // anaglyph shaders
@@ -160,7 +161,7 @@ struct VideoConfig final
bool bSupportsPostProcessing;
bool bSupportsPaletteConversion;
bool bSupportsClipControl; // Needed by VertexShaderGen, so must stay in VideoCommon
- bool bSupportsCopySubImage; // Needed for partial texture updates
+ bool bSupportsSSAA;
} backend_info;
// Utility
diff --git a/Source/Core/VideoCommon/XFMemory.cpp b/Source/Core/VideoCommon/XFMemory.cpp
index 0e54fea373..1e9e71c5b8 100644
--- a/Source/Core/VideoCommon/XFMemory.cpp
+++ b/Source/Core/VideoCommon/XFMemory.cpp
@@ -2,6 +2,7 @@
// Licensed under GPLv2+
// Refer to the license.txt file included.
+#include "Common/Common.h"
#include "VideoCommon/XFMemory.h"
// STATE_TO_SAVE
diff --git a/Source/Core/VideoCommon/XFMemory.h b/Source/Core/VideoCommon/XFMemory.h
index 29a3aa9a65..86a1ce89a1 100644
--- a/Source/Core/VideoCommon/XFMemory.h
+++ b/Source/Core/VideoCommon/XFMemory.h
@@ -8,92 +8,126 @@
#include "VideoCommon/CPMemory.h"
#include "VideoCommon/DataReader.h"
-
// Lighting
+// Projection
+enum : u32
+{
+ XF_TEXPROJ_ST = 0,
+ XF_TEXPROJ_STQ = 1
+};
-#define XF_TEXPROJ_ST 0
-#define XF_TEXPROJ_STQ 1
-
-#define XF_TEXINPUT_AB11 0
-#define XF_TEXINPUT_ABC1 1
+// Input form
+enum : u32
+{
+ XF_TEXINPUT_AB11 = 0,
+ XF_TEXINPUT_ABC1 = 1
+};
-#define XF_TEXGEN_REGULAR 0
-#define XF_TEXGEN_EMBOSS_MAP 1 // used when bump mapping
-#define XF_TEXGEN_COLOR_STRGBC0 2
-#define XF_TEXGEN_COLOR_STRGBC1 3
+// Texture generation type
+enum : u32
+{
+ XF_TEXGEN_REGULAR = 0,
+ XF_TEXGEN_EMBOSS_MAP = 1, // Used when bump mapping
+ XF_TEXGEN_COLOR_STRGBC0 = 2,
+ XF_TEXGEN_COLOR_STRGBC1 = 3
+};
-#define XF_SRCGEOM_INROW 0 // input is abc
-#define XF_SRCNORMAL_INROW 1 // input is abc
-#define XF_SRCCOLORS_INROW 2
-#define XF_SRCBINORMAL_T_INROW 3 // input is abc
-#define XF_SRCBINORMAL_B_INROW 4 // input is abc
-#define XF_SRCTEX0_INROW 5
-#define XF_SRCTEX1_INROW 6
-#define XF_SRCTEX2_INROW 7
-#define XF_SRCTEX3_INROW 8
-#define XF_SRCTEX4_INROW 9
-#define XF_SRCTEX5_INROW 10
-#define XF_SRCTEX6_INROW 11
-#define XF_SRCTEX7_INROW 12
+// Source row
+enum : u32
+{
+ XF_SRCGEOM_INROW = 0, // Input is abc
+ XF_SRCNORMAL_INROW = 1, // Input is abc
+ XF_SRCCOLORS_INROW = 2,
+ XF_SRCBINORMAL_T_INROW = 3, // Input is abc
+ XF_SRCBINORMAL_B_INROW = 4, // Input is abc
+ XF_SRCTEX0_INROW = 5,
+ XF_SRCTEX1_INROW = 6,
+ XF_SRCTEX2_INROW = 7,
+ XF_SRCTEX3_INROW = 8,
+ XF_SRCTEX4_INROW = 9,
+ XF_SRCTEX5_INROW = 10,
+ XF_SRCTEX6_INROW = 11,
+ XF_SRCTEX7_INROW = 12
+};
-#define GX_SRC_REG 0
-#define GX_SRC_VTX 1
+// Control source
+enum : u32
+{
+ GX_SRC_REG = 0,
+ GX_SRC_VTX = 1
+};
-#define LIGHTDIF_NONE 0
-#define LIGHTDIF_SIGN 1
-#define LIGHTDIF_CLAMP 2
+// Light diffuse attenuation function
+enum : u32
+{
+ LIGHTDIF_NONE = 0,
+ LIGHTDIF_SIGN = 1,
+ LIGHTDIF_CLAMP = 2
+};
-#define LIGHTATTN_NONE 0 // no attenuation
-#define LIGHTATTN_SPEC 1 // point light attenuation
-#define LIGHTATTN_DIR 2 // directional light attenuation
-#define LIGHTATTN_SPOT 3 // spot light attenuation
+// Light attenuation function
+enum : u32
+{
+ LIGHTATTN_NONE = 0, // No attenuation
+ LIGHTATTN_SPEC = 1, // Point light attenuation
+ LIGHTATTN_DIR = 2, // Directional light attenuation
+ LIGHTATTN_SPOT = 3 // Spot light attenuation
+};
-#define GX_PERSPECTIVE 0
-#define GX_ORTHOGRAPHIC 1
+// Projection type
+enum : u32
+{
+ GX_PERSPECTIVE = 0,
+ GX_ORTHOGRAPHIC = 1
+};
-#define XFMEM_POSMATRICES 0x000
-#define XFMEM_POSMATRICES_END 0x100
-#define XFMEM_NORMALMATRICES 0x400
-#define XFMEM_NORMALMATRICES_END 0x460
-#define XFMEM_POSTMATRICES 0x500
-#define XFMEM_POSTMATRICES_END 0x600
-#define XFMEM_LIGHTS 0x600
-#define XFMEM_LIGHTS_END 0x680
-#define XFMEM_ERROR 0x1000
-#define XFMEM_DIAG 0x1001
-#define XFMEM_STATE0 0x1002
-#define XFMEM_STATE1 0x1003
-#define XFMEM_CLOCK 0x1004
-#define XFMEM_CLIPDISABLE 0x1005
-#define XFMEM_SETGPMETRIC 0x1006
-#define XFMEM_VTXSPECS 0x1008
-#define XFMEM_SETNUMCHAN 0x1009
-#define XFMEM_SETCHAN0_AMBCOLOR 0x100a
-#define XFMEM_SETCHAN1_AMBCOLOR 0x100b
-#define XFMEM_SETCHAN0_MATCOLOR 0x100c
-#define XFMEM_SETCHAN1_MATCOLOR 0x100d
-#define XFMEM_SETCHAN0_COLOR 0x100e
-#define XFMEM_SETCHAN1_COLOR 0x100f
-#define XFMEM_SETCHAN0_ALPHA 0x1010
-#define XFMEM_SETCHAN1_ALPHA 0x1011
-#define XFMEM_DUALTEX 0x1012
-#define XFMEM_SETMATRIXINDA 0x1018
-#define XFMEM_SETMATRIXINDB 0x1019
-#define XFMEM_SETVIEWPORT 0x101a
-#define XFMEM_SETZSCALE 0x101c
-#define XFMEM_SETZOFFSET 0x101f
-#define XFMEM_SETPROJECTION 0x1020
-/*#define XFMEM_SETPROJECTIONB 0x1021
-#define XFMEM_SETPROJECTIONC 0x1022
-#define XFMEM_SETPROJECTIOND 0x1023
-#define XFMEM_SETPROJECTIONE 0x1024
-#define XFMEM_SETPROJECTIONF 0x1025
-#define XFMEM_SETPROJECTIONORTHO1 0x1026
-#define XFMEM_SETPROJECTIONORTHO2 0x1027*/
-#define XFMEM_SETNUMTEXGENS 0x103f
-#define XFMEM_SETTEXMTXINFO 0x1040
-#define XFMEM_SETPOSMTXINFO 0x1050
+// Registers and register ranges
+enum
+{
+ XFMEM_POSMATRICES = 0x000,
+ XFMEM_POSMATRICES_END = 0x100,
+ XFMEM_NORMALMATRICES = 0x400,
+ XFMEM_NORMALMATRICES_END = 0x460,
+ XFMEM_POSTMATRICES = 0x500,
+ XFMEM_POSTMATRICES_END = 0x600,
+ XFMEM_LIGHTS = 0x600,
+ XFMEM_LIGHTS_END = 0x680,
+ XFMEM_ERROR = 0x1000,
+ XFMEM_DIAG = 0x1001,
+ XFMEM_STATE0 = 0x1002,
+ XFMEM_STATE1 = 0x1003,
+ XFMEM_CLOCK = 0x1004,
+ XFMEM_CLIPDISABLE = 0x1005,
+ XFMEM_SETGPMETRIC = 0x1006,
+ XFMEM_VTXSPECS = 0x1008,
+ XFMEM_SETNUMCHAN = 0x1009,
+ XFMEM_SETCHAN0_AMBCOLOR = 0x100a,
+ XFMEM_SETCHAN1_AMBCOLOR = 0x100b,
+ XFMEM_SETCHAN0_MATCOLOR = 0x100c,
+ XFMEM_SETCHAN1_MATCOLOR = 0x100d,
+ XFMEM_SETCHAN0_COLOR = 0x100e,
+ XFMEM_SETCHAN1_COLOR = 0x100f,
+ XFMEM_SETCHAN0_ALPHA = 0x1010,
+ XFMEM_SETCHAN1_ALPHA = 0x1011,
+ XFMEM_DUALTEX = 0x1012,
+ XFMEM_SETMATRIXINDA = 0x1018,
+ XFMEM_SETMATRIXINDB = 0x1019,
+ XFMEM_SETVIEWPORT = 0x101a,
+ XFMEM_SETZSCALE = 0x101c,
+ XFMEM_SETZOFFSET = 0x101f,
+ XFMEM_SETPROJECTION = 0x1020,
+ // XFMEM_SETPROJECTIONB = 0x1021,
+ // XFMEM_SETPROJECTIONC = 0x1022,
+ // XFMEM_SETPROJECTIOND = 0x1023,
+ // XFMEM_SETPROJECTIONE = 0x1024,
+ // XFMEM_SETPROJECTIONF = 0x1025,
+ // XFMEM_SETPROJECTIONORTHO1 = 0x1026,
+ // XFMEM_SETPROJECTIONORTHO2 = 0x1027,
+ XFMEM_SETNUMTEXGENS = 0x103f,
+ XFMEM_SETTEXMTXINFO = 0x1040,
+ XFMEM_SETPOSMTXINFO = 0x1050,
+};
union LitChannel
{
diff --git a/Source/Core/VideoCommon/XFStructs.cpp b/Source/Core/VideoCommon/XFStructs.cpp
index 089a458683..6a517dc0ab 100644
--- a/Source/Core/VideoCommon/XFStructs.cpp
+++ b/Source/Core/VideoCommon/XFStructs.cpp
@@ -62,7 +62,7 @@ static void XFRegWritten(int transferSize, u32 baseAddress, DataReader src)
if (xfmem.ambColor[chan] != newValue)
{
VertexManager::Flush();
- VertexShaderManager::SetMaterialColorChanged(chan, newValue);
+ VertexShaderManager::SetMaterialColorChanged(chan);
}
break;
}
@@ -74,7 +74,7 @@ static void XFRegWritten(int transferSize, u32 baseAddress, DataReader src)
if (xfmem.matColor[chan] != newValue)
{
VertexManager::Flush();
- VertexShaderManager::SetMaterialColorChanged(chan + 2, newValue);
+ VertexShaderManager::SetMaterialColorChanged(chan + 2);
}
break;
}