summaryrefslogtreecommitdiff
path: root/Source/Core/VideoCommon
diff options
context:
space:
mode:
Diffstat (limited to 'Source/Core/VideoCommon')
-rw-r--r--Source/Core/VideoCommon/SConscript48
-rw-r--r--Source/Core/VideoCommon/Src/AVIDump.cpp1
-rw-r--r--Source/Core/VideoCommon/Src/BPMemory.cpp110
-rw-r--r--Source/Core/VideoCommon/Src/BPMemory.h4
-rw-r--r--Source/Core/VideoCommon/Src/BPStructs.cpp11
-rw-r--r--Source/Core/VideoCommon/Src/CommandProcessor.cpp116
-rw-r--r--Source/Core/VideoCommon/Src/CommandProcessor.h6
-rw-r--r--Source/Core/VideoCommon/Src/Fifo.cpp4
-rw-r--r--Source/Core/VideoCommon/Src/MainBase.cpp5
-rw-r--r--Source/Core/VideoCommon/Src/OpenCL/OCLTextureDecoder.cpp6
-rw-r--r--Source/Core/VideoCommon/Src/PixelEngine.cpp28
-rw-r--r--Source/Core/VideoCommon/Src/PixelEngine.h3
-rw-r--r--Source/Core/VideoCommon/Src/PixelShaderGen.cpp219
-rw-r--r--Source/Core/VideoCommon/Src/PixelShaderGen.h19
-rw-r--r--Source/Core/VideoCommon/Src/PixelShaderManager.cpp42
-rw-r--r--Source/Core/VideoCommon/Src/PixelShaderManager.h6
-rw-r--r--Source/Core/VideoCommon/Src/RenderBase.cpp2
-rw-r--r--Source/Core/VideoCommon/Src/TextureCacheBase.cpp18
-rw-r--r--Source/Core/VideoCommon/Src/TextureConversionShader.cpp2
-rw-r--r--Source/Core/VideoCommon/Src/VertexShaderGen.cpp73
-rw-r--r--Source/Core/VideoCommon/Src/VertexShaderGen.h2
-rw-r--r--Source/Core/VideoCommon/Src/VertexShaderManager.cpp6
-rw-r--r--Source/Core/VideoCommon/VideoCommon.vcxproj1
-rw-r--r--Source/Core/VideoCommon/VideoCommon.vcxproj.filters1
24 files changed, 404 insertions, 329 deletions
diff --git a/Source/Core/VideoCommon/SConscript b/Source/Core/VideoCommon/SConscript
deleted file mode 100644
index 5499f422db..0000000000
--- a/Source/Core/VideoCommon/SConscript
+++ /dev/null
@@ -1,48 +0,0 @@
-# -*- python -*-
-
-Import('env')
-
-files = [
- 'Src/BPFunctions.cpp',
- 'Src/BPMemory.cpp',
- 'Src/BPStructs.cpp',
- 'Src/CPMemory.cpp',
- 'Src/CommandProcessor.cpp',
- 'Src/DLCache.cpp',
- 'Src/Debugger.cpp',
- 'Src/Fifo.cpp',
- 'Src/FramebufferManagerBase.cpp',
- 'Src/HiresTextures.cpp',
- 'Src/ImageWrite.cpp',
- 'Src/IndexGenerator.cpp',
- 'Src/LightingShaderGen.cpp',
- 'Src/MainBase.cpp',
- 'Src/OnScreenDisplay.cpp',
- 'Src/OpcodeDecoding.cpp',
- 'Src/OpenCL.cpp',
- 'Src/OpenCL/OCLTextureDecoder.cpp',
- 'Src/PixelEngine.cpp',
- 'Src/PixelShaderGen.cpp',
- 'Src/PixelShaderManager.cpp',
- 'Src/RenderBase.cpp',
- 'Src/Statistics.cpp',
- 'Src/TextureCacheBase.cpp',
- 'Src/TextureConversionShader.cpp',
- 'Src/TextureDecoder.cpp',
- 'Src/VertexLoader.cpp',
- 'Src/VertexLoaderManager.cpp',
- 'Src/VertexLoader_Color.cpp',
- 'Src/VertexLoader_Normal.cpp',
- 'Src/VertexLoader_Position.cpp',
- 'Src/VertexLoader_TextCoord.cpp',
- 'Src/VertexManagerBase.cpp',
- 'Src/VertexShaderGen.cpp',
- 'Src/VertexShaderManager.cpp',
- 'Src/VideoConfig.cpp',
- 'Src/VideoState.cpp',
- 'Src/XFMemory.cpp',
- 'Src/XFStructs.cpp',
- 'Src/memcpy_amd.cpp',
- ]
-
-env['LIBS'] += env.StaticLibrary('videocommon', files)
diff --git a/Source/Core/VideoCommon/Src/AVIDump.cpp b/Source/Core/VideoCommon/Src/AVIDump.cpp
index 670693690b..4c31798d9f 100644
--- a/Source/Core/VideoCommon/Src/AVIDump.cpp
+++ b/Source/Core/VideoCommon/Src/AVIDump.cpp
@@ -211,6 +211,7 @@ extern "C" {
#include <libavcodec/avcodec.h>
#include <libavformat/avformat.h>
#include <libswscale/swscale.h>
+#include <libavutil/mathematics.h>
}
AVFormatContext *s_FormatContext = NULL;
diff --git a/Source/Core/VideoCommon/Src/BPMemory.cpp b/Source/Core/VideoCommon/Src/BPMemory.cpp
index 5c4ed457e3..d5b762c1a4 100644
--- a/Source/Core/VideoCommon/Src/BPMemory.cpp
+++ b/Source/Core/VideoCommon/Src/BPMemory.cpp
@@ -66,3 +66,113 @@ void BPReload()
}
}
}
+
+
+void GetBPRegInfo(const u8* data, char* name, size_t name_size, char* desc, size_t desc_size)
+{
+ const char* no_yes[2] = { "No", "Yes" };
+
+ u32 cmddata = Common::swap32(*(u32*)data) & 0xFFFFFF;
+ switch (data[0])
+ {
+ // Macro to set the register name and make sure it was written correctly via compile time assertion
+ #define SetRegName(reg) \
+ snprintf(name, name_size, #reg); \
+ (void)(reg);
+
+ case BPMEM_DISPLAYCOPYFILER: // 0x01
+ // TODO: This is actually the sample pattern used for copies from an antialiased EFB
+ SetRegName(BPMEM_DISPLAYCOPYFILER);
+ // TODO: Description
+ break;
+
+ case 0x02: // 0x02
+ case 0x03: // 0x03
+ case 0x04: // 0x04
+ // TODO: same as BPMEM_DISPLAYCOPYFILER
+ break;
+
+ case BPMEM_EFB_TL: // 0x49
+ {
+ SetRegName(BPMEM_EFB_TL);
+ X10Y10 left_top; left_top.hex = cmddata;
+ snprintf(desc, desc_size, "Left: %d\nTop: %d", left_top.x, left_top.y);
+ }
+ break;
+
+ case BPMEM_EFB_BR: // 0x4A
+ {
+ // TODO: Misleading name, should be BPMEM_EFB_WH instead
+ SetRegName(BPMEM_EFB_BR);
+ X10Y10 width_height; width_height.hex = cmddata;
+ snprintf(desc, desc_size, "Width: %d\nHeight: %d", width_height.x+1, width_height.y+1);
+ }
+ break;
+
+ case BPMEM_EFB_ADDR: // 0x4B
+ SetRegName(BPMEM_EFB_ADDR);
+ snprintf(desc, desc_size, "Target address (32 byte aligned): 0x%06X", cmddata << 5);
+ break;
+
+ case BPMEM_COPYYSCALE: // 0x4E
+ SetRegName(BPMEM_COPYYSCALE);
+ snprintf(desc, desc_size, "Scaling factor (XFB copy only): 0x%X (%f or inverted %f)", cmddata, (float)cmddata/256.f, 256.f/(float)cmddata);
+ break;
+
+ case BPMEM_CLEAR_AR: // 0x4F
+ SetRegName(BPMEM_CLEAR_AR);
+ snprintf(desc, desc_size, "Alpha: 0x%02X\nRed: 0x%02X", (cmddata&0xFF00)>>8, cmddata&0xFF);
+ break;
+
+ case BPMEM_CLEAR_GB: // 0x50
+ SetRegName(BPMEM_CLEAR_GB);
+ snprintf(desc, desc_size, "Green: 0x%02X\nBlue: 0x%02X", (cmddata&0xFF00)>>8, cmddata&0xFF);
+ break;
+
+ case BPMEM_CLEAR_Z: // 0x51
+ SetRegName(BPMEM_CLEAR_Z);
+ snprintf(desc, desc_size, "Z value: 0x%06X", cmddata);
+ break;
+
+ case BPMEM_TRIGGER_EFB_COPY: // 0x52
+ {
+ SetRegName(BPMEM_TRIGGER_EFB_COPY);
+ UPE_Copy copy; copy.Hex = cmddata;
+ snprintf(desc, desc_size, "Clamping: %s\n"
+ "Converting from RGB to YUV: %s\n"
+ "Target pixel format: 0x%X\n"
+ "Gamma correction: %s\n"
+ "Mipmap filter: %s\n"
+ "Vertical scaling: %s\n"
+ "Clear: %s\n"
+ "Frame to field: 0x%01X\n"
+ "Copy to XFB: %s\n"
+ "Intensity format: %s\n"
+ "Automatic color conversion: %s",
+ (copy.clamp0 && copy.clamp1) ? "Top and Bottom" : (copy.clamp0) ? "Top only" : (copy.clamp1) ? "Bottom only" : "None",
+ no_yes[copy.yuv],
+ copy.tp_realFormat(),
+ (copy.gamma==0)?"1.0":(copy.gamma==1)?"1.7":(copy.gamma==2)?"2.2":"Invalid value 0x3?",
+ no_yes[copy.half_scale],
+ no_yes[copy.scale_invert],
+ no_yes[copy.clear],
+ copy.frame_to_field,
+ no_yes[copy.copy_to_xfb],
+ no_yes[copy.intensity_fmt],
+ no_yes[copy.auto_conv]);
+ }
+ break;
+
+ case BPMEM_COPYFILTER0: // 0x53
+ SetRegName(BPMEM_COPYFILTER0);
+ // TODO: Description
+ break;
+
+ case BPMEM_COPYFILTER1: // 0x54
+ SetRegName(BPMEM_COPYFILTER1);
+ // TODO: Description
+ break;
+
+#undef SET_REG_NAME
+ }
+}
diff --git a/Source/Core/VideoCommon/Src/BPMemory.h b/Source/Core/VideoCommon/Src/BPMemory.h
index 0bb2bf9013..9284674aa1 100644
--- a/Source/Core/VideoCommon/Src/BPMemory.h
+++ b/Source/Core/VideoCommon/Src/BPMemory.h
@@ -62,7 +62,7 @@
#define BPMEM_COPYFILTER1 0x54
#define BPMEM_CLEARBBOX1 0x55
#define BPMEM_CLEARBBOX2 0x56
-#define BPMEM_UNKOWN_57 0x57
+#define BPMEM_UNKNOWN_57 0x57
#define BPMEM_REVBITS 0x58
#define BPMEM_SCISSOROFFSET 0x59
#define BPMEM_PRELOAD_ADDR 0x60
@@ -995,4 +995,6 @@ extern BPMemory bpmem;
void LoadBPReg(u32 value0);
+void GetBPRegInfo(const u8* data, char* name, size_t name_size, char* desc, size_t desc_size);
+
#endif // _BPMEMORY_H
diff --git a/Source/Core/VideoCommon/Src/BPStructs.cpp b/Source/Core/VideoCommon/Src/BPStructs.cpp
index 6cb102555b..d59523e117 100644
--- a/Source/Core/VideoCommon/Src/BPStructs.cpp
+++ b/Source/Core/VideoCommon/Src/BPStructs.cpp
@@ -463,8 +463,8 @@ void BPWritten(const BPCmd& bp)
case BPMEM_REVBITS: // Always set to 0x0F when GX_InitRevBits() is called.
break;
- case BPMEM_UNKOWN_57: // Sunshine alternates this register between values 0x000 and 0xAAA
- DEBUG_LOG(VIDEO, "Uknown BP Reg 0x57: %08x", bp.newvalue);
+ case BPMEM_UNKNOWN_57: // Sunshine alternates this register between values 0x000 and 0xAAA
+ DEBUG_LOG(VIDEO, "Unknown BP Reg 0x57: %08x", bp.newvalue);
break;
case BPMEM_PRELOAD_ADDR:
@@ -481,6 +481,13 @@ void BPWritten(const BPCmd& bp)
u8* ram_ptr = Memory::GetPointer(tmem_cfg.preload_addr << 5);
u32 tmem_addr = tmem_cfg.preload_tmem_even * TMEM_LINE_SIZE;
u32 size = tmem_cfg.preload_tile_info.count * 32;
+
+ // Check if the game has overflowed TMEM, and copy up to the limit.
+ // Paper Mario does this when entering the Great Boogly Tree (Chap 2)
+ // TODO: Does this wrap?
+ if ((tmem_addr + size) > TMEM_SIZE)
+ size = TMEM_SIZE - tmem_addr;
+
memcpy(texMem + tmem_addr, ram_ptr, size);
}
break;
diff --git a/Source/Core/VideoCommon/Src/CommandProcessor.cpp b/Source/Core/VideoCommon/Src/CommandProcessor.cpp
index fc85b8e39d..a4dddf537f 100644
--- a/Source/Core/VideoCommon/Src/CommandProcessor.cpp
+++ b/Source/Core/VideoCommon/Src/CommandProcessor.cpp
@@ -56,11 +56,12 @@ static bool bProcessFifoToLoWatermark = false;
static bool bProcessFifoAllDistance = false;
volatile bool isPossibleWaitingSetDrawDone = false;
+volatile bool isHiWatermarkActive = false;
volatile bool interruptSet= false;
volatile bool interruptWaiting= false;
volatile bool interruptTokenWaiting = false;
volatile bool interruptFinishWaiting = false;
-volatile bool OnOverflow = false;
+volatile bool waitingForPEInterruptDisable = false;
bool IsOnThread()
{
@@ -86,13 +87,12 @@ void DoState(PointerWrap &p)
p.Do(bProcessFifoToLoWatermark);
p.Do(bProcessFifoAllDistance);
-
+ p.Do(isHiWatermarkActive);
p.Do(isPossibleWaitingSetDrawDone);
p.Do(interruptSet);
p.Do(interruptWaiting);
p.Do(interruptTokenWaiting);
p.Do(interruptFinishWaiting);
- p.Do(OnOverflow);
}
inline void WriteLow (volatile u32& _reg, u16 lowbits) {Common::AtomicStore(_reg,(_reg & 0xFFFF0000) | lowbits);}
@@ -135,16 +135,14 @@ void Init()
bProcessFifoToLoWatermark = false;
bProcessFifoAllDistance = false;
isPossibleWaitingSetDrawDone = false;
- OnOverflow = false;
+ isHiWatermarkActive = false;
et_UpdateInterrupts = CoreTiming::RegisterEvent("UpdateInterrupts", UpdateInterrupts_Wrapper);
}
void Read16(u16& _rReturnValue, const u32 _Address)
{
-
INFO_LOG(COMMANDPROCESSOR, "(r): 0x%08x", _Address);
- ProcessFifoEvents();
switch (_Address & 0xFFF)
{
case STATUS_REGISTER:
@@ -173,11 +171,23 @@ void Read16(u16& _rReturnValue, const u32 _Address)
case FIFO_LO_WATERMARK_HI: _rReturnValue = ReadHigh(fifo.CPLoWatermark); return;
case FIFO_RW_DISTANCE_LO:
- _rReturnValue = ReadLow (fifo.CPReadWriteDistance);
+ if (IsOnThread())
+ if(fifo.CPWritePointer >= fifo.SafeCPReadPointer)
+ _rReturnValue = ReadLow (fifo.CPWritePointer - fifo.SafeCPReadPointer);
+ else
+ _rReturnValue = ReadLow (fifo.CPEnd - fifo.SafeCPReadPointer + fifo.CPWritePointer - fifo.CPBase + 32);
+ else
+ _rReturnValue = ReadLow (fifo.CPReadWriteDistance);
DEBUG_LOG(COMMANDPROCESSOR, "read FIFO_RW_DISTANCE_LO : %04x", _rReturnValue);
return;
case FIFO_RW_DISTANCE_HI:
- _rReturnValue = ReadHigh(fifo.CPReadWriteDistance);
+ if (IsOnThread())
+ if(fifo.CPWritePointer >= fifo.SafeCPReadPointer)
+ _rReturnValue = ReadHigh (fifo.CPWritePointer - fifo.SafeCPReadPointer);
+ else
+ _rReturnValue = ReadHigh (fifo.CPEnd - fifo.SafeCPReadPointer + fifo.CPWritePointer - fifo.CPBase + 32);
+ else
+ _rReturnValue = ReadHigh(fifo.CPReadWriteDistance);
DEBUG_LOG(COMMANDPROCESSOR, "read FIFO_RW_DISTANCE_HI : %04x", _rReturnValue);
return;
case FIFO_WRITE_POINTER_LO:
@@ -358,6 +368,7 @@ void Write16(const u16 _Value, const u32 _Address)
break;
case FIFO_READ_POINTER_HI:
WriteHigh((u32 &)fifo.CPReadPointer, _Value);
+ fifo.SafeCPReadPointer = fifo.CPReadPointer;
DEBUG_LOG(COMMANDPROCESSOR,"\t write to FIFO_READ_POINTER_HI : %04x", _Value);
break;
@@ -390,10 +401,6 @@ void Write16(const u16 _Value, const u32 _Address)
case FIFO_RW_DISTANCE_HI:
WriteHigh((u32 &)fifo.CPReadWriteDistance, _Value);
- DEBUG_LOG(COMMANDPROCESSOR,"try to write to FIFO_RW_DISTANCE_HI : %04x", _Value);
- break;
- case FIFO_RW_DISTANCE_LO:
- WriteLow((u32 &)fifo.CPReadWriteDistance, _Value & 0xFFE0);
if (fifo.CPReadWriteDistance == 0)
{
GPFifo::ResetGatherPipe();
@@ -403,6 +410,10 @@ void Write16(const u16 _Value, const u32 _Address)
ResetVideoBuffer();
}
IncrementCheckContextId();
+ DEBUG_LOG(COMMANDPROCESSOR,"try to write to FIFO_RW_DISTANCE_HI : %04x", _Value);
+ break;
+ case FIFO_RW_DISTANCE_LO:
+ WriteLow((u32 &)fifo.CPReadWriteDistance, _Value & 0xFFE0);
DEBUG_LOG(COMMANDPROCESSOR,"try to write to FIFO_RW_DISTANCE_LO : %04x", _Value);
break;
@@ -412,7 +423,6 @@ void Write16(const u16 _Value, const u32 _Address)
if (!IsOnThread())
RunGpu();
- ProcessFifoEvents();
}
void Read32(u32& _rReturnValue, const u32 _Address)
@@ -434,6 +444,19 @@ void STACKALIGN GatherPipeBursted()
{
if (!IsOnThread())
RunGpu();
+ else
+ {
+ // In multibuffer mode is not allowed write in the same fifo attached to the GPU.
+ // Fix Pokemon XD in DC mode.
+ if((ProcessorInterface::Fifo_CPUEnd == fifo.CPEnd) && (ProcessorInterface::Fifo_CPUBase == fifo.CPBase)
+ && fifo.CPReadWriteDistance > 0)
+ {
+ waitingForPEInterruptDisable = true;
+ ProcessFifoAllDistance();
+ waitingForPEInterruptDisable = false;
+ }
+
+ }
return;
}
@@ -449,26 +472,7 @@ void STACKALIGN GatherPipeBursted()
Common::AtomicAdd(fifo.CPReadWriteDistance, GATHER_PIPE_SIZE);
if (!IsOnThread())
- {
RunGpu();
- }
- else
- {
- if(fifo.CPReadWriteDistance == fifo.CPEnd - fifo.CPBase - 32)
- {
- if(!OnOverflow)
- NOTICE_LOG(COMMANDPROCESSOR,"FIFO is almost in overflown, BreakPoint: %i", fifo.bFF_Breakpoint);
- OnOverflow = true;
- while (!CommandProcessor::interruptWaiting && fifo.bFF_GPReadEnable &&
- fifo.CPReadWriteDistance > fifo.CPEnd - fifo.CPBase - 64)
- Common::YieldCPU();
- }
- else
- {
- OnOverflow = false;
- }
- }
-
_assert_msg_(COMMANDPROCESSOR, fifo.CPReadWriteDistance <= fifo.CPEnd - fifo.CPBase,
"FIFO is overflown by GatherPipe !\nCPU thread is too fast!");
@@ -509,17 +513,15 @@ void AbortFrame()
void SetOverflowStatusFromGatherPipe()
{
- if (!fifo.bFF_HiWatermarkInt) return;
-
fifo.bFF_HiWatermark = (fifo.CPReadWriteDistance > fifo.CPHiWatermark);
- fifo.bFF_LoWatermark = (fifo.CPReadWriteDistance < fifo.CPLoWatermark);
-
- bool interrupt = fifo.bFF_HiWatermark && fifo.bFF_HiWatermarkInt &&
- m_CPCtrlReg.GPLinkEnable && m_CPCtrlReg.GPReadEnable;
+ isHiWatermarkActive = fifo.bFF_HiWatermark && fifo.bFF_HiWatermarkInt && m_CPCtrlReg.GPReadEnable;
- if (interrupt != interruptSet && interrupt)
- CommandProcessor::UpdateInterrupts(true);
-
+ if (isHiWatermarkActive)
+ {
+ interruptSet = true;
+ INFO_LOG(COMMANDPROCESSOR,"Interrupt set");
+ ProcessorInterface::SetInterrupt(INT_CAUSE_CP, true);
+ }
}
void SetCpStatus()
@@ -527,14 +529,12 @@ void SetCpStatus()
// overflow & underflow check
fifo.bFF_HiWatermark = (fifo.CPReadWriteDistance > fifo.CPHiWatermark);
fifo.bFF_LoWatermark = (fifo.CPReadWriteDistance < fifo.CPLoWatermark);
-
- // breakpoint
+ // breakpoint
if (fifo.bFF_BPEnable)
{
if (fifo.CPBreakpoint == fifo.CPReadPointer)
- {
-
+ {
if (!fifo.bFF_Breakpoint)
{
INFO_LOG(COMMANDPROCESSOR, "Hit breakpoint at %i", fifo.CPReadPointer);
@@ -562,13 +562,18 @@ void SetCpStatus()
bool interrupt = (bpInt || ovfInt || undfInt) && m_CPCtrlReg.GPReadEnable;
+ isHiWatermarkActive = ovfInt && m_CPCtrlReg.GPReadEnable;
+
if (interrupt != interruptSet && !interruptWaiting)
{
u64 userdata = interrupt?1:0;
if (IsOnThread())
{
- interruptWaiting = true;
- CommandProcessor::UpdateInterruptsFromVideoBackend(userdata);
+ if(!interrupt || bpInt || undfInt)
+ {
+ interruptWaiting = true;
+ CommandProcessor::UpdateInterruptsFromVideoBackend(userdata);
+ }
}
else
CommandProcessor::UpdateInterrupts(userdata);
@@ -591,7 +596,7 @@ void ProcessFifoAllDistance()
if (IsOnThread())
{
while (!CommandProcessor::interruptWaiting && fifo.bFF_GPReadEnable &&
- fifo.CPReadWriteDistance && !AtBreakpoint())
+ fifo.CPReadWriteDistance && !AtBreakpoint() && !PixelEngine::WaitingForPEInterrupt())
Common::YieldCPU();
}
bProcessFifoAllDistance = false;
@@ -611,13 +616,16 @@ void Shutdown()
void SetCpStatusRegister()
{
// Here always there is one fifo attached to the GPU
-
m_CPStatusReg.Breakpoint = fifo.bFF_Breakpoint;
- m_CPStatusReg.ReadIdle = (fifo.CPReadPointer == fifo.CPWritePointer) || (fifo.CPReadPointer == fifo.CPBreakpoint);
+ m_CPStatusReg.ReadIdle = !fifo.CPReadWriteDistance || (fifo.CPReadPointer == fifo.CPWritePointer) || (fifo.CPReadPointer == fifo.CPBreakpoint) ;
m_CPStatusReg.CommandIdle = !fifo.CPReadWriteDistance;
m_CPStatusReg.UnderflowLoWatermark = fifo.bFF_LoWatermark;
m_CPStatusReg.OverflowHiWatermark = fifo.bFF_HiWatermark;
+ // HACK to compensate for slow response to PE interrupts in Time Splitters: Future Perfect
+ if (IsOnThread())
+ PixelEngine::ResumeWaitingForPEInterrupt();
+
INFO_LOG(COMMANDPROCESSOR,"\t Read from STATUS_REGISTER : %04x", m_CPStatusReg.Hex);
DEBUG_LOG(COMMANDPROCESSOR, "(r) status: iBP %s | fReadIdle %s | fCmdIdle %s | iOvF %s | iUndF %s"
, m_CPStatusReg.Breakpoint ? "ON" : "OFF"
@@ -630,14 +638,14 @@ void SetCpStatusRegister()
void SetCpControlRegister()
{
-
// If the new fifo is being attached We make sure there wont be SetFinish event pending.
// This protection fix eternal darkness booting, because the second SetFinish event when it is booting
// seems invalid or has a bug and hang the game.
if (!fifo.bFF_GPReadEnable && m_CPCtrlReg.GPReadEnable && !m_CPCtrlReg.BPEnable)
{
- PixelEngine::ResetSetFinish();
+ ProcessFifoEvents();
+ PixelEngine::ResetSetFinish();
}
fifo.bFF_BPInt = m_CPCtrlReg.BPInt;
@@ -652,9 +660,6 @@ void SetCpControlRegister()
ProcessorInterface::Fifo_CPUBase = fifo.CPBase;
ProcessorInterface::Fifo_CPUEnd = fifo.CPEnd;
}
- // If overflown happens process the fifo to LoWatemark
- if (bProcessFifoToLoWatermark)
- ProcessFifoToLoWatermark();
if(fifo.bFF_GPReadEnable && !m_CPCtrlReg.GPReadEnable)
{
@@ -666,7 +671,6 @@ void SetCpControlRegister()
fifo.bFF_GPReadEnable = m_CPCtrlReg.GPReadEnable;
}
-
DEBUG_LOG(COMMANDPROCESSOR, "\t GPREAD %s | BP %s | Int %s | OvF %s | UndF %s | LINK %s"
, fifo.bFF_GPReadEnable ? "ON" : "OFF"
, fifo.bFF_BPEnable ? "ON" : "OFF"
diff --git a/Source/Core/VideoCommon/Src/CommandProcessor.h b/Source/Core/VideoCommon/Src/CommandProcessor.h
index db6772d66d..5d31453537 100644
--- a/Source/Core/VideoCommon/Src/CommandProcessor.h
+++ b/Source/Core/VideoCommon/Src/CommandProcessor.h
@@ -25,18 +25,18 @@ class PointerWrap;
extern bool MT;
-
namespace CommandProcessor
{
extern SCPFifoStruct fifo; //This one is shared between gfx thread and emulator thread.
extern volatile bool isPossibleWaitingSetDrawDone; //This one is used for sync gfx thread and emulator thread.
+extern volatile bool isHiWatermarkActive;
extern volatile bool interruptSet;
extern volatile bool interruptWaiting;
extern volatile bool interruptTokenWaiting;
extern volatile bool interruptFinishWaiting;
-extern volatile bool OnOverflow;
-
+extern volatile bool waitingForPEInterruptDisable;
+
// internal hardware addresses
enum
{
diff --git a/Source/Core/VideoCommon/Src/Fifo.cpp b/Source/Core/VideoCommon/Src/Fifo.cpp
index 842ff49e78..2e60055c97 100644
--- a/Source/Core/VideoCommon/Src/Fifo.cpp
+++ b/Source/Core/VideoCommon/Src/Fifo.cpp
@@ -22,6 +22,7 @@
#include "Atomic.h"
#include "OpcodeDecoding.h"
#include "CommandProcessor.h"
+#include "PixelEngine.h"
#include "ChunkFile.h"
#include "Fifo.h"
#include "HW/Memmap.h"
@@ -137,8 +138,7 @@ void RunGpuLoop()
CommandProcessor::SetCpStatus();
// check if we are able to run this buffer
- while (!CommandProcessor::interruptWaiting && fifo.bFF_GPReadEnable &&
- fifo.CPReadWriteDistance && (!AtBreakpoint() || CommandProcessor::OnOverflow))
+ while (GpuRunningState && !CommandProcessor::interruptWaiting && fifo.bFF_GPReadEnable && fifo.CPReadWriteDistance && !AtBreakpoint() && !PixelEngine::WaitingForPEInterrupt())
{
if (!GpuRunningState) break;
diff --git a/Source/Core/VideoCommon/Src/MainBase.cpp b/Source/Core/VideoCommon/Src/MainBase.cpp
index af21ebbb94..0b1258a662 100644
--- a/Source/Core/VideoCommon/Src/MainBase.cpp
+++ b/Source/Core/VideoCommon/Src/MainBase.cpp
@@ -250,6 +250,11 @@ bool VideoBackendHardware::Video_IsPossibleWaitingSetDrawDone()
return CommandProcessor::isPossibleWaitingSetDrawDone;
}
+bool VideoBackendHardware::Video_IsHiWatermarkActive()
+{
+ return CommandProcessor::isHiWatermarkActive;
+}
+
void VideoBackendHardware::Video_AbortFrame()
{
CommandProcessor::AbortFrame();
diff --git a/Source/Core/VideoCommon/Src/OpenCL/OCLTextureDecoder.cpp b/Source/Core/VideoCommon/Src/OpenCL/OCLTextureDecoder.cpp
index 0f4809a6cc..440e69efd6 100644
--- a/Source/Core/VideoCommon/Src/OpenCL/OCLTextureDecoder.cpp
+++ b/Source/Core/VideoCommon/Src/OpenCL/OCLTextureDecoder.cpp
@@ -114,7 +114,7 @@ void TexDecoder_OpenCL_Initialize()
else
{
binary_size = input.GetSize();
- header = new char[HEADER_SIZE]; // TODO: memleak possible
+ header = new char[HEADER_SIZE];
binary = new char[binary_size];
input.ReadBytes(header, HEADER_SIZE);
input.ReadBytes(binary, binary_size);
@@ -143,8 +143,8 @@ void TexDecoder_OpenCL_Initialize()
}
}
}
- delete header;
- delete binary;
+ delete [] header;
+ delete [] binary;
}
// If an error occurred using the kernel binary, recompile the kernels
diff --git a/Source/Core/VideoCommon/Src/PixelEngine.cpp b/Source/Core/VideoCommon/Src/PixelEngine.cpp
index 456e0fb535..03a3a7547a 100644
--- a/Source/Core/VideoCommon/Src/PixelEngine.cpp
+++ b/Source/Core/VideoCommon/Src/PixelEngine.cpp
@@ -180,7 +180,6 @@ void Init()
void Read16(u16& _uReturnValue, const u32 _iAddress)
{
DEBUG_LOG(PIXELENGINE, "(r16) 0x%08x", _iAddress);
- CommandProcessor::ProcessFifoEvents();
switch (_iAddress & 0xFFF)
{
// CPU Direct Access EFB Raster State Config
@@ -269,6 +268,10 @@ void Read16(u16& _uReturnValue, const u32 _iAddress)
case PE_PERF_5L:
case PE_PERF_5H:
INFO_LOG(PIXELENGINE, "(r16) perf counter @ %08x", _iAddress);
+ // git r90a2096a24f4 (svn r3663) added the PE_PERF cases, without setting
+ // _uReturnValue to anything, this reverts to the previous behaviour which allows
+ // The timer in SMS:Scrubbing Serena Beach to countdown correctly
+ _uReturnValue = 1;
break;
default:
@@ -323,7 +326,6 @@ void Write16(const u16 _iValue, const u32 _iAddress)
break;
case PE_TOKEN_REG:
- //LOG(PIXELENGINE,"WEIRD: program wrote token: %i",_iValue);
PanicAlert("(w16) WTF? PowerPC program wrote token: %i", _iValue);
//only the gx pipeline is supposed to be able to write here
//g_token = _iValue;
@@ -334,7 +336,6 @@ void Write16(const u16 _iValue, const u32 _iAddress)
break;
}
- CommandProcessor::ProcessFifoEvents();
}
void Write32(const u32 _iValue, const u32 _iAddress)
@@ -358,22 +359,16 @@ void UpdateInterrupts()
void UpdateTokenInterrupt(bool active)
{
- if(interruptSetToken != active)
- {
ProcessorInterface::SetInterrupt(INT_CAUSE_PE_TOKEN, active);
interruptSetToken = active;
- }
}
void UpdateFinishInterrupt(bool active)
{
- if(interruptSetFinish != active)
- {
ProcessorInterface::SetInterrupt(INT_CAUSE_PE_FINISH, active);
interruptSetFinish = active;
if (active)
State::ProcessRequestedStates(0);
- }
}
// TODO(mb2): Refactor SetTokenINT_OnMainThread(u64 userdata, int cyclesLate).
@@ -392,8 +387,6 @@ void SetToken_OnMainThread(u64 userdata, int cyclesLate)
CommandProcessor::interruptTokenWaiting = false;
IncrementCheckContextId();
//}
- //else
- // LOGV(PIXELENGINE, 1, "VIDEO Backend wrote token: %i", CommandProcessor::fifo.PEToken);
}
void SetFinish_OnMainThread(u64 userdata, int cyclesLate)
@@ -470,4 +463,17 @@ void ResetSetToken()
}
CommandProcessor::interruptTokenWaiting = false;
}
+
+bool WaitingForPEInterrupt()
+{
+ return !CommandProcessor::waitingForPEInterruptDisable && (CommandProcessor::interruptFinishWaiting || CommandProcessor::interruptTokenWaiting || interruptSetFinish || interruptSetToken);
+}
+
+void ResumeWaitingForPEInterrupt()
+{
+ interruptSetFinish = false;
+ interruptSetToken = false;
+ CommandProcessor::interruptFinishWaiting = false;
+ CommandProcessor::interruptTokenWaiting = false;
+}
} // end of namespace PixelEngine
diff --git a/Source/Core/VideoCommon/Src/PixelEngine.h b/Source/Core/VideoCommon/Src/PixelEngine.h
index dd0304fbe1..64f959009f 100644
--- a/Source/Core/VideoCommon/Src/PixelEngine.h
+++ b/Source/Core/VideoCommon/Src/PixelEngine.h
@@ -80,7 +80,8 @@ void SetToken(const u16 _token, const int _bSetTokenAcknowledge);
void SetFinish(void);
void ResetSetFinish(void);
void ResetSetToken(void);
-bool AllowIdleSkipping();
+bool WaitingForPEInterrupt();
+void ResumeWaitingForPEInterrupt();
// Bounding box functionality. Paper Mario (both) are a couple of the few games that use it.
extern u16 bbox[4];
diff --git a/Source/Core/VideoCommon/Src/PixelShaderGen.cpp b/Source/Core/VideoCommon/Src/PixelShaderGen.cpp
index fe5827de05..7500997fef 100644
--- a/Source/Core/VideoCommon/Src/PixelShaderGen.cpp
+++ b/Source/Core/VideoCommon/Src/PixelShaderGen.cpp
@@ -158,7 +158,7 @@ void GetPixelShaderId(PIXELSHADERUID *uid, DSTALPHA_MODE dstAlphaMode, u32 compo
}
u32* ptr = &uid->values[2];
- for (int i = 0; i < bpmem.genMode.numtevstages+1; ++i)
+ for (unsigned int i = 0; i < bpmem.genMode.numtevstages+1; ++i)
{
StageHash(i, ptr);
ptr += 4; // max: ptr = &uid->values[66]
@@ -299,7 +299,7 @@ void ValidatePixelShaderIDs(API_TYPE api, PIXELSHADERUIDSAFE old_id, const std::
static void WriteStage(char *&p, int n, API_TYPE ApiType);
static void SampleTexture(char *&p, const char *destination, const char *texcoords, const char *texswap, int texmap, API_TYPE ApiType);
// static void WriteAlphaCompare(char *&p, int num, int comp);
-static bool WriteAlphaTest(char *&p, API_TYPE ApiType,DSTALPHA_MODE dstAlphaMode);
+static void WriteAlphaTest(char *&p, API_TYPE ApiType,DSTALPHA_MODE dstAlphaMode);
static void WriteFog(char *&p);
static const char *tevKSelTableC[] = // KCSEL
@@ -556,26 +556,22 @@ const char *GeneratePixelShaderCode(DSTALPHA_MODE dstAlphaMode, API_TYPE ApiType
WRITE(p, "\n");
- WRITE(p, "uniform float4 "I_COLORS"[4] : register(c%d);\n", C_COLORS);
- WRITE(p, "uniform float4 "I_KCOLORS"[4] : register(c%d);\n", C_KCOLORS);
- WRITE(p, "uniform float4 "I_ALPHA"[1] : register(c%d);\n", C_ALPHA);
- WRITE(p, "uniform float4 "I_TEXDIMS"[8] : register(c%d);\n", C_TEXDIMS);
- if (ApiType & API_D3D9)
- {
- WRITE(p, "uniform float4 "I_VTEXSCALE"[4] : register(c%d);\n", C_VTEXSCALE);
- }
- WRITE(p, "uniform float4 "I_ZBIAS"[2] : register(c%d);\n", C_ZBIAS);
- WRITE(p, "uniform float4 "I_INDTEXSCALE"[2] : register(c%d);\n", C_INDTEXSCALE);
- WRITE(p, "uniform float4 "I_INDTEXMTX"[6] : register(c%d);\n", C_INDTEXMTX);
- WRITE(p, "uniform float4 "I_FOG"[3] : register(c%d);\n", C_FOG);
+ WRITE(p, "uniform float4 " I_COLORS"[4] : register(c%d);\n", C_COLORS);
+ WRITE(p, "uniform float4 " I_KCOLORS"[4] : register(c%d);\n", C_KCOLORS);
+ WRITE(p, "uniform float4 " I_ALPHA"[1] : register(c%d);\n", C_ALPHA);
+ WRITE(p, "uniform float4 " I_TEXDIMS"[8] : register(c%d);\n", C_TEXDIMS);
+ WRITE(p, "uniform float4 " I_ZBIAS"[2] : register(c%d);\n", C_ZBIAS);
+ WRITE(p, "uniform float4 " I_INDTEXSCALE"[2] : register(c%d);\n", C_INDTEXSCALE);
+ WRITE(p, "uniform float4 " I_INDTEXMTX"[6] : register(c%d);\n", C_INDTEXMTX);
+ WRITE(p, "uniform float4 " I_FOG"[3] : register(c%d);\n", C_FOG);
if(g_ActiveConfig.bEnablePixelLighting && g_ActiveConfig.backend_info.bSupportsPixelLighting)
{
WRITE(p,"typedef struct { float4 col; float4 cosatt; float4 distatt; float4 pos; float4 dir; } Light;\n");
- WRITE(p,"typedef struct { Light lights[8]; } s_"I_PLIGHTS";\n");
- WRITE(p, "uniform s_"I_PLIGHTS" "I_PLIGHTS" : register(c%d);\n", C_PLIGHTS);
- WRITE(p, "typedef struct { float4 C0, C1, C2, C3; } s_"I_PMATERIALS";\n");
- WRITE(p, "uniform s_"I_PMATERIALS" "I_PMATERIALS" : register(c%d);\n", C_PMATERIALS);
+ WRITE(p,"typedef struct { Light lights[8]; } s_" I_PLIGHTS";\n");
+ WRITE(p, "uniform s_" I_PLIGHTS" " I_PLIGHTS" : register(c%d);\n", C_PLIGHTS);
+ WRITE(p, "typedef struct { float4 C0, C1, C2, C3; } s_" I_PMATERIALS";\n");
+ WRITE(p, "uniform s_" I_PMATERIALS" " I_PMATERIALS" : register(c%d);\n", C_PMATERIALS);
}
WRITE(p, "void main(\n");
@@ -623,25 +619,32 @@ const char *GeneratePixelShaderCode(DSTALPHA_MODE dstAlphaMode, API_TYPE ApiType
char* pmainstart = p;
int Pretest = AlphaPreTest();
- if (dstAlphaMode == DSTALPHA_ALPHA_PASS && !DepthTextureEnable && Pretest >= 0)
+ if(Pretest >= 0 && !DepthTextureEnable)
{
if (!Pretest)
{
- // alpha test will always fail, so restart the shader and just make it an empty function
+ // alpha test will always fail, so restart the shader and just make it an empty function
WRITE(p, "ocol0 = 0;\n");
+ if(DepthTextureEnable)
+ WRITE(p, "depth = 1.f;\n");
+ if(dstAlphaMode == DSTALPHA_DUAL_SOURCE_BLEND)
+ WRITE(p, "ocol1 = 0;\n");
WRITE(p, "discard;\n");
if(ApiType != API_D3D11)
WRITE(p, "return;\n");
}
- else
+ else if (dstAlphaMode == DSTALPHA_ALPHA_PASS)
{
- WRITE(p, " ocol0 = "I_ALPHA"[0].aaaa;\n");
+ WRITE(p, " ocol0 = " I_ALPHA"[0].aaaa;\n");
+ }
+ if(!Pretest || dstAlphaMode == DSTALPHA_ALPHA_PASS)
+ {
+ WRITE(p, "}\n");
+ return text;
}
- WRITE(p, "}\n");
- return text;
}
- WRITE(p, " float4 c0 = "I_COLORS"[1], c1 = "I_COLORS"[2], c2 = "I_COLORS"[3], prev = float4(0.0f, 0.0f, 0.0f, 0.0f), textemp = float4(0.0f, 0.0f, 0.0f, 0.0f), rastemp = float4(0.0f, 0.0f, 0.0f, 0.0f), konsttemp = float4(0.0f, 0.0f, 0.0f, 0.0f);\n"
+ WRITE(p, " float4 c0 = " I_COLORS"[1], c1 = " I_COLORS"[2], c2 = " I_COLORS"[3], prev = float4(0.0f, 0.0f, 0.0f, 0.0f), textemp = float4(0.0f, 0.0f, 0.0f, 0.0f), rastemp = float4(0.0f, 0.0f, 0.0f, 0.0f), konsttemp = float4(0.0f, 0.0f, 0.0f, 0.0f);\n"
" float3 comp16 = float3(1.0f, 255.0f, 0.0f), comp24 = float3(1.0f, 255.0f, 255.0f*255.0f);\n"
" float4 alphabump=float4(0.0f,0.0f,0.0f,0.0f);\n"
" float3 tevcoord=float3(0.0f, 0.0f, 0.0f);\n"
@@ -692,7 +695,7 @@ const char *GeneratePixelShaderCode(DSTALPHA_MODE dstAlphaMode, API_TYPE ApiType
WRITE(p, " uv%d.xy = uv%d.xy / uv%d.z;\n", i, i, i);
}
- WRITE(p, "uv%d.xy = uv%d.xy * "I_TEXDIMS"[%d].zw;\n", i, i, i);
+ WRITE(p, "uv%d.xy = uv%d.xy * " I_TEXDIMS"[%d].zw;\n", i, i, i);
}
}
@@ -704,7 +707,7 @@ const char *GeneratePixelShaderCode(DSTALPHA_MODE dstAlphaMode, API_TYPE ApiType
int texcoord = bpmem.tevindref.getTexCoord(i);
if (texcoord < numTexgen)
- WRITE(p, "tempcoord = uv%d.xy * "I_INDTEXSCALE"[%d].%s;\n", texcoord, i/2, (i&1)?"zw":"xy");
+ WRITE(p, "tempcoord = uv%d.xy * " I_INDTEXSCALE"[%d].%s;\n", texcoord, i/2, (i&1)?"zw":"xy");
else
WRITE(p, "tempcoord = float2(0.0f, 0.0f);\n");
@@ -727,65 +730,53 @@ const char *GeneratePixelShaderCode(DSTALPHA_MODE dstAlphaMode, API_TYPE ApiType
// emulation of unsigned 8 overflow when casting
WRITE(p, "prev = frac(4.0f + prev * (255.0f/256.0f)) * (256.0f/255.0f);\n");
- // TODO: Why are we doing a second alpha pretest here?
- if (!WriteAlphaTest(p, ApiType, dstAlphaMode))
+ if(Pretest == -1)
{
- // alpha test will always fail, so restart the shader and just make it an empty function
- p = pmainstart;
- WRITE(p, "ocol0 = 0;\n");
- if(DepthTextureEnable)
- WRITE(p, "depth = 1.f;\n");
- if(dstAlphaMode == DSTALPHA_DUAL_SOURCE_BLEND)
- WRITE(p, "ocol1 = 0;\n");
- WRITE(p, "discard;\n");
- if(ApiType != API_D3D11)
- WRITE(p, "return;\n");
+ WriteAlphaTest(p, ApiType, dstAlphaMode);
}
- else
- {
- if((bpmem.fog.c_proj_fsel.fsel != 0) || DepthTextureEnable)
- {
- // the screen space depth value = far z + (clip z / clip w) * z range
- WRITE(p, "float zCoord = "I_ZBIAS"[1].x + (clipPos.z / clipPos.w) * "I_ZBIAS"[1].y;\n");
- }
+ if((bpmem.fog.c_proj_fsel.fsel != 0) || DepthTextureEnable)
+ {
+ // the screen space depth value = far z + (clip z / clip w) * z range
+ WRITE(p, "float zCoord = " I_ZBIAS"[1].x + (clipPos.z / clipPos.w) * " I_ZBIAS"[1].y;\n");
+ }
- if (DepthTextureEnable)
+ if (DepthTextureEnable)
+ {
+ // use the texture input of the last texture stage (textemp), hopefully this has been read and is in correct format...
+ if (bpmem.ztex2.op != ZTEXTURE_DISABLE && !bpmem.zcontrol.zcomploc && bpmem.zmode.testenable && bpmem.zmode.updateenable)
{
- // use the texture input of the last texture stage (textemp), hopefully this has been read and is in correct format...
- if (bpmem.ztex2.op != ZTEXTURE_DISABLE && !bpmem.zcontrol.zcomploc && bpmem.zmode.testenable && bpmem.zmode.updateenable)
- {
- if (bpmem.ztex2.op == ZTEXTURE_ADD)
- WRITE(p, "zCoord = dot("I_ZBIAS"[0].xyzw, textemp.xyzw) + "I_ZBIAS"[1].w + zCoord;\n");
- else
- WRITE(p, "zCoord = dot("I_ZBIAS"[0].xyzw, textemp.xyzw) + "I_ZBIAS"[1].w;\n");
-
- // scale to make result from frac correct
- WRITE(p, "zCoord = zCoord * (16777215.0f/16777216.0f);\n");
- WRITE(p, "zCoord = frac(zCoord);\n");
- WRITE(p, "zCoord = zCoord * (16777216.0f/16777215.0f);\n");
- }
- WRITE(p, "depth = zCoord;\n");
- }
+ if (bpmem.ztex2.op == ZTEXTURE_ADD)
+ WRITE(p, "zCoord = dot(" I_ZBIAS"[0].xyzw, textemp.xyzw) + " I_ZBIAS"[1].w + zCoord;\n");
+ else
+ WRITE(p, "zCoord = dot(" I_ZBIAS"[0].xyzw, textemp.xyzw) + " I_ZBIAS"[1].w;\n");
- if (dstAlphaMode == DSTALPHA_ALPHA_PASS)
- WRITE(p, " ocol0 = float4(prev.rgb, "I_ALPHA"[0].a);\n");
- else
- {
- WriteFog(p);
- WRITE(p, " ocol0 = prev;\n");
+ // scale to make result from frac correct
+ WRITE(p, "zCoord = zCoord * (16777215.0f/16777216.0f);\n");
+ WRITE(p, "zCoord = frac(zCoord);\n");
+ WRITE(p, "zCoord = zCoord * (16777216.0f/16777215.0f);\n");
}
+ WRITE(p, "depth = zCoord;\n");
+ }
- // On D3D11, use dual-source color blending to perform dst alpha in a
- // single pass
- if (dstAlphaMode == DSTALPHA_DUAL_SOURCE_BLEND)
- {
- // Colors will be blended against the alpha from ocol1...
- WRITE(p, " ocol1 = ocol0;\n");
- // ...and the alpha from ocol0 will be written to the framebuffer.
- WRITE(p, " ocol0.a = "I_ALPHA"[0].a;\n");
- }
+ if (dstAlphaMode == DSTALPHA_ALPHA_PASS)
+ WRITE(p, " ocol0 = float4(prev.rgb, " I_ALPHA"[0].a);\n");
+ else
+ {
+ WriteFog(p);
+ WRITE(p, " ocol0 = prev;\n");
}
+
+ // On D3D11, use dual-source color blending to perform dst alpha in a
+ // single pass
+ if (dstAlphaMode == DSTALPHA_DUAL_SOURCE_BLEND)
+ {
+ // Colors will be blended against the alpha from ocol1...
+ WRITE(p, " ocol1 = ocol0;\n");
+ // ...and the alpha from ocol0 will be written to the framebuffer.
+ WRITE(p, " ocol0.a = " I_ALPHA"[0].a;\n");
+ }
+
WRITE(p, "}\n");
if (text[sizeof(text) - 1] != 0x7C)
PanicAlert("PixelShader generator - buffer too small, canary has been eaten!");
@@ -876,20 +867,20 @@ static void WriteStage(char *&p, int n, API_TYPE ApiType)
if (bpmem.tevind[n].mid <= 3)
{
int mtxidx = 2*(bpmem.tevind[n].mid-1);
- WRITE(p, "float2 indtevtrans%d = float2(dot("I_INDTEXMTX"[%d].xyz, indtevcrd%d), dot("I_INDTEXMTX"[%d].xyz, indtevcrd%d));\n",
+ WRITE(p, "float2 indtevtrans%d = float2(dot(" I_INDTEXMTX"[%d].xyz, indtevcrd%d), dot(" I_INDTEXMTX"[%d].xyz, indtevcrd%d));\n",
n, mtxidx, n, mtxidx+1, n);
}
else if (bpmem.tevind[n].mid <= 7 && bHasTexCoord)
{ // s matrix
_assert_(bpmem.tevind[n].mid >= 5);
int mtxidx = 2*(bpmem.tevind[n].mid-5);
- WRITE(p, "float2 indtevtrans%d = "I_INDTEXMTX"[%d].ww * uv%d.xy * indtevcrd%d.xx;\n", n, mtxidx, texcoord, n);
+ WRITE(p, "float2 indtevtrans%d = " I_INDTEXMTX"[%d].ww * uv%d.xy * indtevcrd%d.xx;\n", n, mtxidx, texcoord, n);
}
else if (bpmem.tevind[n].mid <= 11 && bHasTexCoord)
{ // t matrix
_assert_(bpmem.tevind[n].mid >= 9);
int mtxidx = 2*(bpmem.tevind[n].mid-9);
- WRITE(p, "float2 indtevtrans%d = "I_INDTEXMTX"[%d].ww * uv%d.xy * indtevcrd%d.yy;\n", n, mtxidx, texcoord, n);
+ WRITE(p, "float2 indtevtrans%d = " I_INDTEXMTX"[%d].ww * uv%d.xy * indtevcrd%d.yy;\n", n, mtxidx, texcoord, n);
}
else
WRITE(p, "float2 indtevtrans%d = 0;\n", n);
@@ -1103,15 +1094,9 @@ static void WriteStage(char *&p, int n, API_TYPE ApiType)
void SampleTexture(char *&p, const char *destination, const char *texcoords, const char *texswap, int texmap, API_TYPE ApiType)
{
if (ApiType == API_D3D11)
- WRITE(p, "%s=Tex%d.Sample(samp%d, %s.xy * "I_TEXDIMS"[%d].xy).%s;\n", destination, texmap,texmap, texcoords, texmap, texswap);
- else if (ApiType & API_D3D9)
- {
- // D3D9 uses different pixel to texel mapping, so we need to offset our sampling address by half a pixel (assuming native and virtual texture dimensions match each other, otherwise some math is involved).
- // Read the MSDN article "Directly Mapping Texels to Pixels (Direct3D 9)" for further info.
- WRITE(p, "%s=tex2D(samp%d, (%s.xy + 0.5f*"I_VTEXSCALE"[%d].%s) * "I_TEXDIMS"[%d].xy).%s;\n", destination, texmap, texcoords, texmap/2, (texmap&1)?"zw":"xy", texmap, texswap);
- }
+ WRITE(p, "%s=Tex%d.Sample(samp%d,%s.xy * " I_TEXDIMS"[%d].xy).%s;\n", destination, texmap,texmap, texcoords, texmap, texswap);
else
- WRITE(p, "%s=tex2D(samp%d, %s.xy * "I_TEXDIMS"[%d].xy).%s;\n", destination, texmap, texcoords, texmap, texswap);
+ WRITE(p, "%s=tex2D(samp%d,%s.xy * " I_TEXDIMS"[%d].xy).%s;\n", destination, texmap, texcoords, texmap, texswap);
}
static const char *tevAlphaFuncsTable[] =
@@ -1167,19 +1152,13 @@ static int AlphaPreTest()
}
-static bool WriteAlphaTest(char *&p, API_TYPE ApiType,DSTALPHA_MODE dstAlphaMode)
+static void WriteAlphaTest(char *&p, API_TYPE ApiType,DSTALPHA_MODE dstAlphaMode)
{
static const char *alphaRef[2] =
{
I_ALPHA"[0].r",
I_ALPHA"[0].g"
- };
-
- int Pretest = AlphaPreTest();
- if(Pretest >= 0)
- {
- return Pretest != 0;
- }
+ };
// using discard then return works the same in cg and dx9 but not in dx11
WRITE(p, "if(!( ");
@@ -1191,11 +1170,35 @@ static bool WriteAlphaTest(char *&p, API_TYPE ApiType,DSTALPHA_MODE dstAlphaMode
compindex = bpmem.alphaFunc.comp1 % 8;
WRITE(p, tevAlphaFuncsTable[compindex],alphaRef[1]);//lookup the second component from the alpha function table
- WRITE(p, ")){ocol0 = 0;%s%s discard;%s}\n",
- dstAlphaMode == DSTALPHA_DUAL_SOURCE_BLEND ? "ocol1 = 0;" : "",
- DepthTextureEnable ? "depth = 1.f;" : "",
- (ApiType != API_D3D11) ? "return;" : "");
- return true;
+ WRITE(p, ")) {\n");
+
+ WRITE(p, "ocol0 = 0;\n");
+ if (dstAlphaMode == DSTALPHA_DUAL_SOURCE_BLEND)
+ WRITE(p, "ocol1 = 0;\n");
+ if (DepthTextureEnable)
+ WRITE(p, "depth = 1.f;\n");
+
+ // HAXX: zcomploc is a way to control whether depth test is done before
+ // or after texturing and alpha test. PC GPU does depth test before texturing ONLY if depth value is
+ // not updated during shader execution.
+ // We implement "depth test before texturing" by discarding the fragment
+ // when the alpha test fail. This is not a correct implementation because
+ // even if the depth test fails the fragment could be alpha blended.
+ // this implemnetation is a trick to keep speed.
+ // the correct, but slow, way to implement a correct zComploc is :
+ // 1 - if zcomplock is enebled make a first pass, with color channel write disabled updating only
+ // depth channel.
+ // 2 - in the next pass disable depth chanel update, but proccess the color data normally
+ // this way is the only CORRECT way to emulate perfectly the zcomplock behaviour
+ if (!(bpmem.zcontrol.zcomploc && bpmem.zmode.updateenable))
+ {
+ WRITE(p, "discard;\n");
+ if (ApiType != API_D3D11)
+ WRITE(p, "return;\n");
+ }
+
+ WRITE(p, "}\n");
+
}
static const char *tevFogFuncsTable[] =
@@ -1218,13 +1221,13 @@ static void WriteFog(char *&p)
{
// perspective
// ze = A/(B - (Zs >> B_SHF)
- WRITE (p, " float ze = "I_FOG"[1].x / ("I_FOG"[1].y - (zCoord / "I_FOG"[1].w));\n");
+ WRITE (p, " float ze = " I_FOG"[1].x / (" I_FOG"[1].y - (zCoord / " I_FOG"[1].w));\n");
}
else
{
// orthographic
// ze = a*Zs (here, no B_SHF)
- WRITE (p, " float ze = "I_FOG"[1].x * zCoord;\n");
+ WRITE (p, " float ze = " I_FOG"[1].x * zCoord;\n");
}
// x_adjust = sqrt((x-center)^2 + k^2)/k
@@ -1232,12 +1235,12 @@ static void WriteFog(char *&p)
//this is complitly teorical as the real hard seems to use a table intead of calculate the values.
if(bpmem.fogRange.Base.Enabled)
{
- WRITE (p, " float x_adjust = (2.0f * (clipPos.x / "I_FOG"[2].y)) - 1.0f - "I_FOG"[2].x;\n");
- WRITE (p, " x_adjust = sqrt(x_adjust * x_adjust + "I_FOG"[2].z * "I_FOG"[2].z) / "I_FOG"[2].z;\n");
+ WRITE (p, " float x_adjust = (2.0f * (clipPos.x / " I_FOG"[2].y)) - 1.0f - " I_FOG"[2].x;\n");
+ WRITE (p, " x_adjust = sqrt(x_adjust * x_adjust + " I_FOG"[2].z * " I_FOG"[2].z) / " I_FOG"[2].z;\n");
WRITE (p, " ze *= x_adjust;\n");
}
- WRITE (p, " float fog = saturate(ze - "I_FOG"[1].z);\n");
+ WRITE (p, " float fog = saturate(ze - " I_FOG"[1].z);\n");
if(bpmem.fog.c_proj_fsel.fsel > 3)
{
@@ -1249,7 +1252,7 @@ static void WriteFog(char *&p)
WARN_LOG(VIDEO, "Unknown Fog Type! %08x", bpmem.fog.c_proj_fsel.fsel);
}
- WRITE(p, " prev.rgb = lerp(prev.rgb,"I_FOG"[0].rgb,fog);\n");
+ WRITE(p, " prev.rgb = lerp(prev.rgb," I_FOG"[0].rgb,fog);\n");
-} \ No newline at end of file
+}
diff --git a/Source/Core/VideoCommon/Src/PixelShaderGen.h b/Source/Core/VideoCommon/Src/PixelShaderGen.h
index 7374251c96..31242a916e 100644
--- a/Source/Core/VideoCommon/Src/PixelShaderGen.h
+++ b/Source/Core/VideoCommon/Src/PixelShaderGen.h
@@ -24,27 +24,26 @@
#define I_KCOLORS "k"
#define I_ALPHA "alphaRef"
#define I_TEXDIMS "texdim"
-#define I_VTEXSCALE "vtexscale"
#define I_ZBIAS "czbias"
#define I_INDTEXSCALE "cindscale"
#define I_INDTEXMTX "cindmtx"
#define I_FOG "cfog"
#define I_PLIGHTS "cLights"
-#define I_PMATERIALS "cmtrl"
+#define I_PMATERIALS "cmtrl"
#define C_COLORMATRIX 0 // 0
#define C_COLORS 0 // 0
#define C_KCOLORS (C_COLORS + 4) // 4
#define C_ALPHA (C_KCOLORS + 4) // 8
#define C_TEXDIMS (C_ALPHA + 1) // 9
-#define C_VTEXSCALE (C_TEXDIMS + 8) //17 - virtual texture scaling factor (e.g. custom textures, scaled EFB copies)
-#define C_ZBIAS (C_VTEXSCALE + 4) //21
-#define C_INDTEXSCALE (C_ZBIAS + 2) //23
-#define C_INDTEXMTX (C_INDTEXSCALE + 2) //25
-#define C_FOG (C_INDTEXMTX + 6) //31
-#define C_PLIGHTS (C_FOG + 3) //34
-#define C_PMATERIALS (C_PLIGHTS + 40) //74
-#define C_PENVCONST_END (C_PMATERIALS + 4) //78
+#define C_ZBIAS (C_TEXDIMS + 8) //17
+#define C_INDTEXSCALE (C_ZBIAS + 2) //19
+#define C_INDTEXMTX (C_INDTEXSCALE + 2) //21
+#define C_FOG (C_INDTEXMTX + 6) //27
+
+#define C_PLIGHTS (C_FOG + 3)
+#define C_PMATERIALS (C_PLIGHTS + 40)
+#define C_PENVCONST_END (C_PMATERIALS + 4)
#define PIXELSHADERUID_MAX_VALUES 70
#define PIXELSHADERUID_MAX_VALUES_SAFE 120
diff --git a/Source/Core/VideoCommon/Src/PixelShaderManager.cpp b/Source/Core/VideoCommon/Src/PixelShaderManager.cpp
index 3e0f6d73c4..2521f80500 100644
--- a/Source/Core/VideoCommon/Src/PixelShaderManager.cpp
+++ b/Source/Core/VideoCommon/Src/PixelShaderManager.cpp
@@ -36,11 +36,9 @@ static bool s_bFogRangeAdjustChanged;
static int nLightsChanged[2]; // min,max
static float lastRGBAfull[2][4][4];
static u8 s_nTexDimsChanged;
-static u8 s_nVirtualTexScalesChanged;
static u8 s_nIndTexScaleChanged;
static u32 lastAlpha;
static u32 lastTexDims[8]; // width | height << 16 | wrap_s << 28 | wrap_t << 30
-static float lastVirtualTexScales[16]; // even fields: width ratio; odd fields: height ratio
static u32 lastZBias;
static int nMaterialsChanged;
@@ -63,7 +61,6 @@ void PixelShaderManager::Init()
{
lastAlpha = 0;
memset(lastTexDims, 0, sizeof(lastTexDims));
- memset(lastVirtualTexScales, 0, sizeof(lastVirtualTexScales));
lastZBias = 0;
memset(lastRGBAfull, 0, sizeof(lastRGBAfull));
Dirty();
@@ -73,7 +70,6 @@ void PixelShaderManager::Dirty()
{
s_nColorsChanged[0] = s_nColorsChanged[1] = 15;
s_nTexDimsChanged = 0xFF;
- s_nVirtualTexScalesChanged = 0xFF;
s_nIndTexScaleChanged = 0xFF;
s_nIndTexMtxChanged = 15;
s_bAlphaChanged = s_bZBiasChanged = s_bZTextureTypeChanged = s_bDepthRangeChanged = true;
@@ -87,7 +83,7 @@ void PixelShaderManager::Shutdown()
}
-void PixelShaderManager::SetConstants(API_TYPE api_type)
+void PixelShaderManager::SetConstants()
{
for (int i = 0; i < 2; ++i)
{
@@ -113,16 +109,6 @@ void PixelShaderManager::SetConstants(API_TYPE api_type)
s_nTexDimsChanged = 0;
}
- if ((api_type & API_D3D9) && s_nVirtualTexScalesChanged)
- {
- for (int i = 0; i < 8; i += 2)
- {
- if (s_nVirtualTexScalesChanged & (3<<i))
- SetPSVirtualTexScalePair(i/2);
- }
- s_nVirtualTexScalesChanged = 0;
- }
-
if (s_bAlphaChanged)
{
SetPSConstant4f(C_ALPHA, (lastAlpha&0xff)/255.0f, ((lastAlpha>>8)&0xff)/255.0f, 0, ((lastAlpha>>16)&0xff)/255.0f);
@@ -352,13 +338,6 @@ void PixelShaderManager::SetPSTextureDims(int texid)
SetPSConstant4fv(C_TEXDIMS + texid, fdims);
}
-void PixelShaderManager::SetPSVirtualTexScalePair(int texpairid)
-{
- PRIM_LOG("vtexscale%d: %f %f %f %f\n", texpairid, lastVirtualTexScales[texpairid*4], lastVirtualTexScales[texpairid*4+1],
- lastVirtualTexScales[texpairid*4+2], lastVirtualTexScales[texpairid*4+3]);
- SetPSConstant4fv(C_VTEXSCALE + texpairid, &lastVirtualTexScales[texpairid*4]);
-}
-
// This one is high in profiles (0.5%). TODO: Move conversion out, only store the raw color value
// and update it when the shader constant is set, only.
void PixelShaderManager::SetColorChanged(int type, int num, bool high)
@@ -397,25 +376,14 @@ void PixelShaderManager::SetDestAlpha(const ConstantAlpha& alpha)
}
}
-void PixelShaderManager::SetTexDims(int texmapid, u32 width, u32 height, u32 virtual_width, u32 virtual_height, u32 wraps, u32 wrapt, API_TYPE api_type)
+void PixelShaderManager::SetTexDims(int texmapid, u32 width, u32 height, u32 wraps, u32 wrapt)
{
u32 wh = width | (height << 16) | (wraps << 28) | (wrapt << 30);
-
- bool refresh = lastTexDims[texmapid] != wh;
- if (api_type & API_D3D9)
+ if (lastTexDims[texmapid] != wh)
{
- refresh |= (lastVirtualTexScales[texmapid*2] != (float)width / (float)virtual_width);
- refresh |= (lastVirtualTexScales[texmapid*2+1] != (float)height / (float)virtual_height);
- }
-
- if (refresh)
- {
- lastTexDims[texmapid] = wh;
- lastVirtualTexScales[texmapid*2] = (float)width / (float)virtual_width;
- lastVirtualTexScales[texmapid*2+1] = (float)height / (float)virtual_height;
+ lastTexDims[texmapid] = wh;
s_nTexDimsChanged |= 1 << texmapid;
- s_nVirtualTexScalesChanged |= 1 << texmapid;
- }
+ }
}
void PixelShaderManager::SetZTextureBias(u32 bias)
diff --git a/Source/Core/VideoCommon/Src/PixelShaderManager.h b/Source/Core/VideoCommon/Src/PixelShaderManager.h
index 5cbe86cae4..2d1c01cad6 100644
--- a/Source/Core/VideoCommon/Src/PixelShaderManager.h
+++ b/Source/Core/VideoCommon/Src/PixelShaderManager.h
@@ -26,20 +26,18 @@
class PixelShaderManager
{
static void SetPSTextureDims(int texid);
- static void SetPSVirtualTexScalePair(int texpairid);
-
public:
static void Init();
static void Dirty();
static void Shutdown();
- static void SetConstants(API_TYPE api_type); // sets pixel shader constants
+ static void SetConstants(); // sets pixel shader constants
// constant management, should be called after memory is committed
static void SetColorChanged(int type, int index, bool high);
static void SetAlpha(const AlphaFunc& alpha);
static void SetDestAlpha(const ConstantAlpha& alpha);
- static void SetTexDims(int texmapid, u32 width, u32 height, u32 virtual_width, u32 virtual_height, u32 wraps, u32 wrapt, API_TYPE api_type);
+ static void SetTexDims(int texmapid, u32 width, u32 height, u32 wraps, u32 wrapt);
static void SetZTextureBias(u32 bias);
static void SetViewportChanged();
static void SetIndMatrixChanged(int matrixidx);
diff --git a/Source/Core/VideoCommon/Src/RenderBase.cpp b/Source/Core/VideoCommon/Src/RenderBase.cpp
index 3b06bbb1c0..ef0fff57f7 100644
--- a/Source/Core/VideoCommon/Src/RenderBase.cpp
+++ b/Source/Core/VideoCommon/Src/RenderBase.cpp
@@ -43,6 +43,7 @@
#include "XFMemory.h"
#include "FifoPlayer/FifoRecorder.h"
#include "AVIDump.h"
+#include "VertexShaderManager.h"
#include <cmath>
#include <string>
@@ -193,6 +194,7 @@ bool Renderer::CalculateTargetSize(int multiplier)
{
s_target_width = newEFBWidth;
s_target_height = newEFBHeight;
+ VertexShaderManager::SetViewportChanged();
return true;
}
return false;
diff --git a/Source/Core/VideoCommon/Src/TextureCacheBase.cpp b/Source/Core/VideoCommon/Src/TextureCacheBase.cpp
index cc14499330..00ff9610be 100644
--- a/Source/Core/VideoCommon/Src/TextureCacheBase.cpp
+++ b/Source/Core/VideoCommon/Src/TextureCacheBase.cpp
@@ -172,8 +172,10 @@ void TextureCache::ClearRenderTargets()
TexCache::iterator
iter = textures.begin(),
tcend = textures.end();
+
for (; iter!=tcend; ++iter)
- iter->second->type = TCET_NORMAL;
+ if (iter->second->type != TCET_EC_DYNAMIC)
+ iter->second->type = TCET_NORMAL;
}
TextureCache::TCacheEntryBase* TextureCache::Load(unsigned int stage,
@@ -238,6 +240,9 @@ TextureCache::TCacheEntryBase* TextureCache::Load(unsigned int stage,
// 2. a) For EFB copies, only the hash and the texture address need to match
if (entry->IsEfbCopy() && tex_hash == entry->hash && address == entry->addr)
{
+ if (entry->type != TCET_EC_VRAM)
+ entry->type = TCET_NORMAL;
+
// TODO: Print a warning if the format changes! In this case, we could reinterpret the internal texture object data to the new pixel format (similiar to what is already being done in Renderer::ReinterpretPixelFormat())
goto return_entry;
}
@@ -318,8 +323,8 @@ TextureCache::TCacheEntryBase* TextureCache::Load(unsigned int stage,
entry->SetGeneralParameters(address, texture_size, full_format, entry->num_mipmaps);
entry->SetDimensions(nativeW, nativeH, width, height);
entry->hash = tex_hash;
- if (g_ActiveConfig.bCopyEFBToTexture) entry->type = TCET_NORMAL;
- else if (entry->IsEfbCopy()) entry->type = TCET_EC_DYNAMIC;
+ if (entry->IsEfbCopy() && !g_ActiveConfig.bCopyEFBToTexture) entry->type = TCET_EC_DYNAMIC;
+ else entry->type = TCET_NORMAL;
// load texture
entry->Load(width, height, expandedWidth, 0, (texLevels == 0));
@@ -647,8 +652,11 @@ void TextureCache::CopyRenderTargetToTexture(u32 dstAddr, unsigned int dstFormat
if ((entry->type == TCET_EC_VRAM && entry->virtual_width == scaled_tex_w && entry->virtual_height == scaled_tex_h)
|| (entry->type == TCET_EC_DYNAMIC && entry->native_width == tex_w && entry->native_height == tex_h))
{
- scaled_tex_w = tex_w;
- scaled_tex_h = tex_h;
+ if (entry->type == TCET_EC_DYNAMIC)
+ {
+ scaled_tex_w = tex_w;
+ scaled_tex_h = tex_h;
+ }
}
else
{
diff --git a/Source/Core/VideoCommon/Src/TextureConversionShader.cpp b/Source/Core/VideoCommon/Src/TextureConversionShader.cpp
index c921a1a9fd..2132a25258 100644
--- a/Source/Core/VideoCommon/Src/TextureConversionShader.cpp
+++ b/Source/Core/VideoCommon/Src/TextureConversionShader.cpp
@@ -259,7 +259,7 @@ void WriteIncrementSampleX(char*& p,API_TYPE ApiType)
void WriteToBitDepth(char*& p, u8 depth, const char* src, const char* dest)
{
- float result = pow(2.0f, depth) - 1.0f;
+ float result = 255 / pow(2.0f, (8 - depth));
WRITE(p, " %s = floor(%s * %ff);\n", dest, src, result);
}
diff --git a/Source/Core/VideoCommon/Src/VertexShaderGen.cpp b/Source/Core/VideoCommon/Src/VertexShaderGen.cpp
index 5076f9f044..07ace04b97 100644
--- a/Source/Core/VideoCommon/Src/VertexShaderGen.cpp
+++ b/Source/Core/VideoCommon/Src/VertexShaderGen.cpp
@@ -178,31 +178,31 @@ const char *GenerateVertexShaderCode(u32 components, API_TYPE api_type)
char *p = text;
WRITE(p, "//Vertex Shader: comp:%x, \n", components);
- WRITE(p, "typedef struct { float4 T0, T1, T2; float4 N0, N1, N2; } s_"I_POSNORMALMATRIX";\n"
+ WRITE(p, "typedef struct { float4 T0, T1, T2; float4 N0, N1, N2; } s_" I_POSNORMALMATRIX";\n"
"typedef struct { float4 t; } FLT4;\n"
- "typedef struct { FLT4 T[24]; } s_"I_TEXMATRICES";\n"
- "typedef struct { FLT4 T[64]; } s_"I_TRANSFORMMATRICES";\n"
- "typedef struct { FLT4 T[32]; } s_"I_NORMALMATRICES";\n"
- "typedef struct { FLT4 T[64]; } s_"I_POSTTRANSFORMMATRICES";\n"
+ "typedef struct { FLT4 T[24]; } s_" I_TEXMATRICES";\n"
+ "typedef struct { FLT4 T[64]; } s_" I_TRANSFORMMATRICES";\n"
+ "typedef struct { FLT4 T[32]; } s_" I_NORMALMATRICES";\n"
+ "typedef struct { FLT4 T[64]; } s_" I_POSTTRANSFORMMATRICES";\n"
"typedef struct { float4 col; float4 cosatt; float4 distatt; float4 pos; float4 dir; } Light;\n"
- "typedef struct { Light lights[8]; } s_"I_LIGHTS";\n"
- "typedef struct { float4 C0, C1, C2, C3; } s_"I_MATERIALS";\n"
- "typedef struct { float4 T0, T1, T2, T3; } s_"I_PROJECTION";\n"
+ "typedef struct { Light lights[8]; } s_" I_LIGHTS";\n"
+ "typedef struct { float4 C0, C1, C2, C3; } s_" I_MATERIALS";\n"
+ "typedef struct { float4 T0, T1, T2, T3; } s_" I_PROJECTION";\n"
);
p = GenerateVSOutputStruct(p, components, api_type);
// uniforms
- WRITE(p, "uniform s_"I_TRANSFORMMATRICES" "I_TRANSFORMMATRICES" : register(c%d);\n", C_TRANSFORMMATRICES);
- WRITE(p, "uniform s_"I_TEXMATRICES" "I_TEXMATRICES" : register(c%d);\n", C_TEXMATRICES); // also using tex matrices
- WRITE(p, "uniform s_"I_NORMALMATRICES" "I_NORMALMATRICES" : register(c%d);\n", C_NORMALMATRICES);
- WRITE(p, "uniform s_"I_POSNORMALMATRIX" "I_POSNORMALMATRIX" : register(c%d);\n", C_POSNORMALMATRIX);
- WRITE(p, "uniform s_"I_POSTTRANSFORMMATRICES" "I_POSTTRANSFORMMATRICES" : register(c%d);\n", C_POSTTRANSFORMMATRICES);
- WRITE(p, "uniform s_"I_LIGHTS" "I_LIGHTS" : register(c%d);\n", C_LIGHTS);
- WRITE(p, "uniform s_"I_MATERIALS" "I_MATERIALS" : register(c%d);\n", C_MATERIALS);
- WRITE(p, "uniform s_"I_PROJECTION" "I_PROJECTION" : register(c%d);\n", C_PROJECTION);
- WRITE(p, "uniform float4 "I_DEPTHPARAMS" : register(c%d);\n", C_DEPTHPARAMS);
+ WRITE(p, "uniform s_" I_TRANSFORMMATRICES" " I_TRANSFORMMATRICES" : register(c%d);\n", C_TRANSFORMMATRICES);
+ WRITE(p, "uniform s_" I_TEXMATRICES" " I_TEXMATRICES" : register(c%d);\n", C_TEXMATRICES); // also using tex matrices
+ WRITE(p, "uniform s_" I_NORMALMATRICES" " I_NORMALMATRICES" : register(c%d);\n", C_NORMALMATRICES);
+ WRITE(p, "uniform s_" I_POSNORMALMATRIX" " I_POSNORMALMATRIX" : register(c%d);\n", C_POSNORMALMATRIX);
+ WRITE(p, "uniform s_" I_POSTTRANSFORMMATRICES" " I_POSTTRANSFORMMATRICES" : register(c%d);\n", C_POSTTRANSFORMMATRICES);
+ WRITE(p, "uniform s_" I_LIGHTS" " I_LIGHTS" : register(c%d);\n", C_LIGHTS);
+ WRITE(p, "uniform s_" I_MATERIALS" " I_MATERIALS" : register(c%d);\n", C_MATERIALS);
+ WRITE(p, "uniform s_" I_PROJECTION" " I_PROJECTION" : register(c%d);\n", C_PROJECTION);
+ WRITE(p, "uniform float4 " I_DEPTHPARAMS" : register(c%d);\n", C_DEPTHPARAMS);
WRITE(p, "VS_OUTPUT main(\n");
@@ -258,11 +258,11 @@ const char *GenerateVertexShaderCode(u32 components, API_TYPE api_type)
WRITE(p, "int posmtx = fposmtx;\n");
}
- WRITE(p, "float4 pos = float4(dot("I_TRANSFORMMATRICES".T[posmtx].t, rawpos), dot("I_TRANSFORMMATRICES".T[posmtx+1].t, rawpos), dot("I_TRANSFORMMATRICES".T[posmtx+2].t, rawpos), 1);\n");
+ WRITE(p, "float4 pos = float4(dot(" I_TRANSFORMMATRICES".T[posmtx].t, rawpos), dot(" I_TRANSFORMMATRICES".T[posmtx+1].t, rawpos), dot(" I_TRANSFORMMATRICES".T[posmtx+2].t, rawpos), 1);\n");
if (components & VB_HAS_NRMALL) {
WRITE(p, "int normidx = posmtx >= 32 ? (posmtx-32) : posmtx;\n");
- WRITE(p, "float3 N0 = "I_NORMALMATRICES".T[normidx].t.xyz, N1 = "I_NORMALMATRICES".T[normidx+1].t.xyz, N2 = "I_NORMALMATRICES".T[normidx+2].t.xyz;\n");
+ WRITE(p, "float3 N0 = " I_NORMALMATRICES".T[normidx].t.xyz, N1 = " I_NORMALMATRICES".T[normidx+1].t.xyz, N2 = " I_NORMALMATRICES".T[normidx+2].t.xyz;\n");
}
if (components & VB_HAS_NRM0)
@@ -274,13 +274,13 @@ const char *GenerateVertexShaderCode(u32 components, API_TYPE api_type)
}
else
{
- WRITE(p, "float4 pos = float4(dot("I_POSNORMALMATRIX".T0, rawpos), dot("I_POSNORMALMATRIX".T1, rawpos), dot("I_POSNORMALMATRIX".T2, rawpos), 1.0f);\n");
+ WRITE(p, "float4 pos = float4(dot(" I_POSNORMALMATRIX".T0, rawpos), dot(" I_POSNORMALMATRIX".T1, rawpos), dot(" I_POSNORMALMATRIX".T2, rawpos), 1.0f);\n");
if (components & VB_HAS_NRM0)
- WRITE(p, "float3 _norm0 = normalize(float3(dot("I_POSNORMALMATRIX".N0.xyz, rawnorm0), dot("I_POSNORMALMATRIX".N1.xyz, rawnorm0), dot("I_POSNORMALMATRIX".N2.xyz, rawnorm0)));\n");
+ WRITE(p, "float3 _norm0 = normalize(float3(dot(" I_POSNORMALMATRIX".N0.xyz, rawnorm0), dot(" I_POSNORMALMATRIX".N1.xyz, rawnorm0), dot(" I_POSNORMALMATRIX".N2.xyz, rawnorm0)));\n");
if (components & VB_HAS_NRM1)
- WRITE(p, "float3 _norm1 = float3(dot("I_POSNORMALMATRIX".N0.xyz, rawnorm1), dot("I_POSNORMALMATRIX".N1.xyz, rawnorm1), dot("I_POSNORMALMATRIX".N2.xyz, rawnorm1));\n");
+ WRITE(p, "float3 _norm1 = float3(dot(" I_POSNORMALMATRIX".N0.xyz, rawnorm1), dot(" I_POSNORMALMATRIX".N1.xyz, rawnorm1), dot(" I_POSNORMALMATRIX".N2.xyz, rawnorm1));\n");
if (components & VB_HAS_NRM2)
- WRITE(p, "float3 _norm2 = float3(dot("I_POSNORMALMATRIX".N0.xyz, rawnorm2), dot("I_POSNORMALMATRIX".N1.xyz, rawnorm2), dot("I_POSNORMALMATRIX".N2.xyz, rawnorm2));\n");
+ WRITE(p, "float3 _norm2 = float3(dot(" I_POSNORMALMATRIX".N0.xyz, rawnorm2), dot(" I_POSNORMALMATRIX".N1.xyz, rawnorm2), dot(" I_POSNORMALMATRIX".N2.xyz, rawnorm2));\n");
}
if (!(components & VB_HAS_NRM0))
@@ -288,7 +288,7 @@ const char *GenerateVertexShaderCode(u32 components, API_TYPE api_type)
- WRITE(p, "o.pos = float4(dot("I_PROJECTION".T0, pos), dot("I_PROJECTION".T1, pos), dot("I_PROJECTION".T2, pos), dot("I_PROJECTION".T3, pos));\n");
+ WRITE(p, "o.pos = float4(dot(" I_PROJECTION".T0, pos), dot(" I_PROJECTION".T1, pos), dot(" I_PROJECTION".T2, pos), dot(" I_PROJECTION".T3, pos));\n");
WRITE(p, "float4 mat, lacc;\n"
"float3 ldir, h;\n"
@@ -367,7 +367,7 @@ const char *GenerateVertexShaderCode(u32 components, API_TYPE api_type)
if (components & (VB_HAS_NRM1|VB_HAS_NRM2)) {
// transform the light dir into tangent space
- WRITE(p, "ldir = normalize("I_LIGHTS".lights[%d].pos.xyz - pos.xyz);\n", texinfo.embosslightshift);
+ WRITE(p, "ldir = normalize(" I_LIGHTS".lights[%d].pos.xyz - pos.xyz);\n", texinfo.embosslightshift);
WRITE(p, "o.tex%d.xyz = o.tex%d.xyz + float3(dot(ldir, _norm1), dot(ldir, _norm2), 0.0f);\n", i, texinfo.embosssourceshift);
}
else
@@ -389,16 +389,16 @@ const char *GenerateVertexShaderCode(u32 components, API_TYPE api_type)
default:
if (components & (VB_HAS_TEXMTXIDX0<<i)) {
if (texinfo.projection == XF_TEXPROJ_STQ)
- WRITE(p, "o.tex%d.xyz = float3(dot(coord, "I_TRANSFORMMATRICES".T[tex%d.z].t), dot(coord, "I_TRANSFORMMATRICES".T[tex%d.z+1].t), dot(coord, "I_TRANSFORMMATRICES".T[tex%d.z+2].t));\n", i, i, i, i);
+ WRITE(p, "o.tex%d.xyz = float3(dot(coord, " I_TRANSFORMMATRICES".T[tex%d.z].t), dot(coord, " I_TRANSFORMMATRICES".T[tex%d.z+1].t), dot(coord, " I_TRANSFORMMATRICES".T[tex%d.z+2].t));\n", i, i, i, i);
else {
- WRITE(p, "o.tex%d.xyz = float3(dot(coord, "I_TRANSFORMMATRICES".T[tex%d.z].t), dot(coord, "I_TRANSFORMMATRICES".T[tex%d.z+1].t), 1);\n", i, i, i);
+ WRITE(p, "o.tex%d.xyz = float3(dot(coord, " I_TRANSFORMMATRICES".T[tex%d.z].t), dot(coord, " I_TRANSFORMMATRICES".T[tex%d.z+1].t), 1);\n", i, i, i);
}
}
else {
if (texinfo.projection == XF_TEXPROJ_STQ)
- WRITE(p, "o.tex%d.xyz = float3(dot(coord, "I_TEXMATRICES".T[%d].t), dot(coord, "I_TEXMATRICES".T[%d].t), dot(coord, "I_TEXMATRICES".T[%d].t));\n", i, 3*i, 3*i+1, 3*i+2);
+ WRITE(p, "o.tex%d.xyz = float3(dot(coord, " I_TEXMATRICES".T[%d].t), dot(coord, " I_TEXMATRICES".T[%d].t), dot(coord, " I_TEXMATRICES".T[%d].t));\n", i, 3*i, 3*i+1, 3*i+2);
else
- WRITE(p, "o.tex%d.xyz = float3(dot(coord, "I_TEXMATRICES".T[%d].t), dot(coord, "I_TEXMATRICES".T[%d].t), 1);\n", i, 3*i, 3*i+1);
+ WRITE(p, "o.tex%d.xyz = float3(dot(coord, " I_TEXMATRICES".T[%d].t), dot(coord, " I_TEXMATRICES".T[%d].t), 1);\n", i, 3*i, 3*i+1);
}
break;
}
@@ -407,9 +407,9 @@ const char *GenerateVertexShaderCode(u32 components, API_TYPE api_type)
const PostMtxInfo& postInfo = xfregs.postMtxInfo[i];
int postidx = postInfo.index;
- WRITE(p, "float4 P0 = "I_POSTTRANSFORMMATRICES".T[%d].t;\n"
- "float4 P1 = "I_POSTTRANSFORMMATRICES".T[%d].t;\n"
- "float4 P2 = "I_POSTTRANSFORMMATRICES".T[%d].t;\n",
+ WRITE(p, "float4 P0 = " I_POSTTRANSFORMMATRICES".T[%d].t;\n"
+ "float4 P1 = " I_POSTTRANSFORMMATRICES".T[%d].t;\n"
+ "float4 P2 = " I_POSTTRANSFORMMATRICES".T[%d].t;\n",
postidx&0x3f, (postidx+1)&0x3f, (postidx+2)&0x3f);
if (texGenSpecialCase) {
@@ -467,7 +467,7 @@ const char *GenerateVertexShaderCode(u32 components, API_TYPE api_type)
//if not early z culling will improve speed
if (is_d3d)
{
- WRITE(p, "o.pos.z = "I_DEPTHPARAMS".x * o.pos.w + o.pos.z * "I_DEPTHPARAMS".y;\n");
+ WRITE(p, "o.pos.z = " I_DEPTHPARAMS".x * o.pos.w + o.pos.z * " I_DEPTHPARAMS".y;\n");
}
else
{
@@ -493,6 +493,13 @@ const char *GenerateVertexShaderCode(u32 components, API_TYPE api_type)
//seems to get rather complicated
}
+ if (api_type & API_D3D9)
+ {
+ // D3D9 is addressing pixel centers instead of pixel boundaries in clip space.
+ // Thus we need to offset the final position by half a pixel
+ WRITE(p, "o.pos = o.pos + float4(" I_DEPTHPARAMS".z, " I_DEPTHPARAMS".w, 0.f, 0.f);\n");
+ }
+
WRITE(p, "return o;\n}\n");
diff --git a/Source/Core/VideoCommon/Src/VertexShaderGen.h b/Source/Core/VideoCommon/Src/VertexShaderGen.h
index 0c522286b1..cb253a9b6c 100644
--- a/Source/Core/VideoCommon/Src/VertexShaderGen.h
+++ b/Source/Core/VideoCommon/Src/VertexShaderGen.h
@@ -35,7 +35,7 @@
#define I_TRANSFORMMATRICES "ctrmtx"
#define I_NORMALMATRICES "cnmtx"
#define I_POSTTRANSFORMMATRICES "cpostmtx"
-#define I_DEPTHPARAMS "cDepth"
+#define I_DEPTHPARAMS "cDepth" // farZ, zRange, scaled viewport width, scaled viewport height
#define C_POSNORMALMATRIX 0
#define C_PROJECTION (C_POSNORMALMATRIX + 6)
diff --git a/Source/Core/VideoCommon/Src/VertexShaderManager.cpp b/Source/Core/VideoCommon/Src/VertexShaderManager.cpp
index d7d25c5c85..b6afbc61c6 100644
--- a/Source/Core/VideoCommon/Src/VertexShaderManager.cpp
+++ b/Source/Core/VideoCommon/Src/VertexShaderManager.cpp
@@ -306,7 +306,11 @@ void VertexShaderManager::SetConstants()
if (bViewportChanged)
{
bViewportChanged = false;
- SetVSConstant4f(C_DEPTHPARAMS,xfregs.viewport.farZ / 16777216.0f,xfregs.viewport.zRange / 16777216.0f,0.0f,0.0f);
+ SetVSConstant4f(C_DEPTHPARAMS,
+ xfregs.viewport.farZ / 16777216.0f,
+ xfregs.viewport.zRange / 16777216.0f,
+ -1.f / (float)g_renderer->EFBToScaledX((int)ceil(2.0f * xfregs.viewport.wd)),
+ 1.f / (float)g_renderer->EFBToScaledY((int)ceil(-2.0f * xfregs.viewport.ht)));
// This is so implementation-dependent that we can't have it here.
UpdateViewport(s_viewportCorrection);
bProjectionChanged = true;
diff --git a/Source/Core/VideoCommon/VideoCommon.vcxproj b/Source/Core/VideoCommon/VideoCommon.vcxproj
index dd29d29ccb..f53c18cb37 100644
--- a/Source/Core/VideoCommon/VideoCommon.vcxproj
+++ b/Source/Core/VideoCommon/VideoCommon.vcxproj
@@ -266,7 +266,6 @@
<ItemGroup>
<None Include="..\..\..\Data\User\OpenCL\TextureDecoder.cl" />
<None Include="CMakeLists.txt" />
- <None Include="Src\SConscript" />
</ItemGroup>
<ItemGroup>
<ProjectReference Include="..\..\..\Externals\CLRun\clrun\CLRun.vcxproj">
diff --git a/Source/Core/VideoCommon/VideoCommon.vcxproj.filters b/Source/Core/VideoCommon/VideoCommon.vcxproj.filters
index 5ce157a78e..c933fbc939 100644
--- a/Source/Core/VideoCommon/VideoCommon.vcxproj.filters
+++ b/Source/Core/VideoCommon/VideoCommon.vcxproj.filters
@@ -248,7 +248,6 @@
</ClInclude>
</ItemGroup>
<ItemGroup>
- <None Include="Src\SConscript" />
<None Include="CMakeLists.txt" />
<None Include="..\..\..\Data\User\OpenCL\TextureDecoder.cl">
<Filter>Decoding\OpenCL</Filter>