From 36c99df9f5d70521d4e2db50f49cfbed753780ed Mon Sep 17 00:00:00 2001 From: Lioncash Date: Tue, 13 Dec 2016 18:23:42 -0500 Subject: Jit: Move most x86-64-specific code out of JitCommon --- Source/Core/Core/CMakeLists.txt | 7 +- Source/Core/Core/Core.vcxproj | 13 +- Source/Core/Core/Core.vcxproj.filters | 37 +- Source/Core/Core/PowerPC/CachedInterpreter.cpp | 1 + Source/Core/Core/PowerPC/Jit64/Jit.cpp | 2 +- Source/Core/Core/PowerPC/Jit64/Jit.h | 2 +- Source/Core/Core/PowerPC/Jit64/JitRegCache.cpp | 2 +- Source/Core/Core/PowerPC/Jit64/Jit_Integer.cpp | 2 +- Source/Core/Core/PowerPC/Jit64/Jit_LoadStore.cpp | 2 +- .../Core/PowerPC/Jit64/Jit_LoadStoreFloating.cpp | 3 +- .../Core/PowerPC/Jit64/Jit_LoadStorePaired.cpp | 5 +- .../Core/PowerPC/Jit64/Jit_SystemRegisters.cpp | 2 +- .../Core/Core/PowerPC/Jit64Common/BlockCache.cpp | 40 + Source/Core/Core/PowerPC/Jit64Common/BlockCache.h | 14 + .../Core/PowerPC/Jit64Common/Jit64AsmCommon.cpp | 4 +- .../Core/Core/PowerPC/Jit64Common/Jit64AsmCommon.h | 3 +- Source/Core/Core/PowerPC/Jit64Common/Jit64Base.cpp | 164 +++ Source/Core/Core/PowerPC/Jit64Common/Jit64Base.h | 48 + Source/Core/Core/PowerPC/Jit64Common/Jit64Util.cpp | 1114 ++++++++++++++++++++ Source/Core/Core/PowerPC/Jit64Common/Jit64Util.h | 212 ++++ .../Core/PowerPC/Jit64Common/TrampolineCache.cpp | 86 ++ .../Core/PowerPC/Jit64Common/TrampolineCache.h | 23 + Source/Core/Core/PowerPC/Jit64IL/JitIL.h | 1 - Source/Core/Core/PowerPC/JitArm64/Jit.h | 3 + .../Core/Core/PowerPC/JitCommon/JitBackpatch.cpp | 124 --- Source/Core/Core/PowerPC/JitCommon/JitBase.cpp | 47 +- Source/Core/Core/PowerPC/JitCommon/JitBase.h | 33 +- Source/Core/Core/PowerPC/JitCommon/JitCache.cpp | 31 - Source/Core/Core/PowerPC/JitCommon/JitCache.h | 8 - Source/Core/Core/PowerPC/JitCommon/Jit_Util.cpp | 1113 ------------------- Source/Core/Core/PowerPC/JitCommon/Jit_Util.h | 211 ---- .../Core/PowerPC/JitCommon/TrampolineCache.cpp | 85 -- .../Core/Core/PowerPC/JitCommon/TrampolineCache.h | 27 - Source/Core/Core/PowerPC/JitILCommon/JitILBase.h | 2 +- 34 files changed, 1756 insertions(+), 1715 deletions(-) create mode 100644 Source/Core/Core/PowerPC/Jit64Common/BlockCache.cpp create mode 100644 Source/Core/Core/PowerPC/Jit64Common/BlockCache.h create mode 100644 Source/Core/Core/PowerPC/Jit64Common/Jit64Base.cpp create mode 100644 Source/Core/Core/PowerPC/Jit64Common/Jit64Base.h create mode 100644 Source/Core/Core/PowerPC/Jit64Common/Jit64Util.cpp create mode 100644 Source/Core/Core/PowerPC/Jit64Common/Jit64Util.h create mode 100644 Source/Core/Core/PowerPC/Jit64Common/TrampolineCache.cpp create mode 100644 Source/Core/Core/PowerPC/Jit64Common/TrampolineCache.h delete mode 100644 Source/Core/Core/PowerPC/JitCommon/JitBackpatch.cpp delete mode 100644 Source/Core/Core/PowerPC/JitCommon/Jit_Util.cpp delete mode 100644 Source/Core/Core/PowerPC/JitCommon/Jit_Util.h delete mode 100644 Source/Core/Core/PowerPC/JitCommon/TrampolineCache.cpp delete mode 100644 Source/Core/Core/PowerPC/JitCommon/TrampolineCache.h (limited to 'Source/Core') diff --git a/Source/Core/Core/CMakeLists.txt b/Source/Core/Core/CMakeLists.txt index aeb3c4be16..0072c57254 100644 --- a/Source/Core/Core/CMakeLists.txt +++ b/Source/Core/Core/CMakeLists.txt @@ -206,10 +206,11 @@ if(_M_X86) PowerPC/Jit64/Jit_Paired.cpp PowerPC/Jit64/JitRegCache.cpp PowerPC/Jit64/Jit_SystemRegisters.cpp + PowerPC/Jit64Common/BlockCache.cpp PowerPC/Jit64Common/Jit64AsmCommon.cpp - PowerPC/JitCommon/JitBackpatch.cpp - PowerPC/JitCommon/Jit_Util.cpp - PowerPC/JitCommon/TrampolineCache.cpp) + PowerPC/Jit64Common/Jit64Base.cpp + PowerPC/Jit64Common/Jit64Util.cpp + PowerPC/Jit64Common/TrampolineCache.cpp) elseif(_M_ARM_64) set(SRCS ${SRCS} PowerPC/JitArm64/Jit.cpp diff --git a/Source/Core/Core/Core.vcxproj b/Source/Core/Core/Core.vcxproj index 7b1745536d..c851e266d4 100644 --- a/Source/Core/Core/Core.vcxproj +++ b/Source/Core/Core/Core.vcxproj @@ -237,13 +237,14 @@ + + + + - - - @@ -429,12 +430,14 @@ + + + + - - diff --git a/Source/Core/Core/Core.vcxproj.filters b/Source/Core/Core/Core.vcxproj.filters index 4a696ec05c..a42cc100c7 100644 --- a/Source/Core/Core/Core.vcxproj.filters +++ b/Source/Core/Core/Core.vcxproj.filters @@ -651,9 +651,6 @@ PowerPC - - PowerPC\JitCommon - PowerPC\JitCommon @@ -663,9 +660,6 @@ PowerPC\JitCommon - - PowerPC\JitCommon - PowerPC\JitIL @@ -741,12 +735,21 @@ PowerPC - - PowerPC\JitCommon + + PowerPC\Jit64Common PowerPC\Jit64Common + + PowerPC\Jit64Common + + + PowerPC\Jit64Common + + + PowerPC\Jit64Common + IPC HLE %28IOS/Starlet%29\USB @@ -1238,9 +1241,6 @@ PowerPC - - PowerPC\JitCommon - PowerPC\JitCommon @@ -1250,9 +1250,6 @@ PowerPC\JitCommon - - PowerPC\JitCommon - PowerPC\JitIL @@ -1278,9 +1275,21 @@ HW %28Flipper/Hollywood%29\GCKeyboard + + PowerPC\Jit64Common + PowerPC\Jit64Common + + PowerPC\Jit64Common + + + PowerPC\Jit64Common + + + PowerPC\Jit64Common + IPC HLE %28IOS/Starlet%29\USB diff --git a/Source/Core/Core/PowerPC/CachedInterpreter.cpp b/Source/Core/Core/PowerPC/CachedInterpreter.cpp index c57ab1e8d8..0173822147 100644 --- a/Source/Core/Core/PowerPC/CachedInterpreter.cpp +++ b/Source/Core/Core/PowerPC/CachedInterpreter.cpp @@ -10,6 +10,7 @@ #include "Core/HLE/HLE.h" #include "Core/HW/CPU.h" #include "Core/PowerPC/Gekko.h" +#include "Core/PowerPC/Jit64Common/Jit64Util.h" #include "Core/PowerPC/PPCAnalyst.h" #include "Core/PowerPC/PowerPC.h" diff --git a/Source/Core/Core/PowerPC/Jit64/Jit.cpp b/Source/Core/Core/PowerPC/Jit64/Jit.cpp index a24ad301cb..6b88e5ecb2 100644 --- a/Source/Core/Core/PowerPC/Jit64/Jit.cpp +++ b/Source/Core/Core/PowerPC/Jit64/Jit.cpp @@ -26,7 +26,7 @@ #include "Core/PowerPC/Jit64/Jit64_Tables.h" #include "Core/PowerPC/Jit64/JitAsm.h" #include "Core/PowerPC/Jit64/JitRegCache.h" -#include "Core/PowerPC/JitCommon/Jit_Util.h" +#include "Core/PowerPC/Jit64Common/Jit64Util.h" #include "Core/PowerPC/JitInterface.h" #include "Core/PowerPC/PowerPC.h" #include "Core/PowerPC/Profiler.h" diff --git a/Source/Core/Core/PowerPC/Jit64/Jit.h b/Source/Core/Core/PowerPC/Jit64/Jit.h index 9ff580d1a2..588d776fab 100644 --- a/Source/Core/Core/PowerPC/Jit64/Jit.h +++ b/Source/Core/Core/PowerPC/Jit64/Jit.h @@ -23,7 +23,7 @@ #include "Common/x64Emitter.h" #include "Core/PowerPC/Jit64/JitAsm.h" #include "Core/PowerPC/Jit64/JitRegCache.h" -#include "Core/PowerPC/JitCommon/JitBase.h" +#include "Core/PowerPC/Jit64Common/Jit64Base.h" #include "Core/PowerPC/JitCommon/JitCache.h" #include "Core/PowerPC/PPCAnalyst.h" diff --git a/Source/Core/Core/PowerPC/Jit64/JitRegCache.cpp b/Source/Core/Core/PowerPC/Jit64/JitRegCache.cpp index 67baa2f485..c3dc984c5d 100644 --- a/Source/Core/Core/PowerPC/Jit64/JitRegCache.cpp +++ b/Source/Core/Core/PowerPC/Jit64/JitRegCache.cpp @@ -14,7 +14,7 @@ #include "Common/x64Emitter.h" #include "Core/PowerPC/Jit64/Jit.h" #include "Core/PowerPC/Jit64/JitRegCache.h" -#include "Core/PowerPC/JitCommon/Jit_Util.h" +#include "Core/PowerPC/Jit64Common/Jit64Util.h" #include "Core/PowerPC/PowerPC.h" using namespace Gen; diff --git a/Source/Core/Core/PowerPC/Jit64/Jit_Integer.cpp b/Source/Core/Core/PowerPC/Jit64/Jit_Integer.cpp index 4b6f262226..a8cd8a24d4 100644 --- a/Source/Core/Core/PowerPC/Jit64/Jit_Integer.cpp +++ b/Source/Core/Core/PowerPC/Jit64/Jit_Integer.cpp @@ -11,7 +11,7 @@ #include "Common/x64Emitter.h" #include "Core/PowerPC/Jit64/Jit.h" #include "Core/PowerPC/Jit64/JitRegCache.h" -#include "Core/PowerPC/JitCommon/Jit_Util.h" +#include "Core/PowerPC/Jit64Common/Jit64Util.h" #include "Core/PowerPC/PPCAnalyst.h" #include "Core/PowerPC/PowerPC.h" diff --git a/Source/Core/Core/PowerPC/Jit64/Jit_LoadStore.cpp b/Source/Core/Core/PowerPC/Jit64/Jit_LoadStore.cpp index c3af9345bf..973cc856cd 100644 --- a/Source/Core/Core/PowerPC/Jit64/Jit_LoadStore.cpp +++ b/Source/Core/Core/PowerPC/Jit64/Jit_LoadStore.cpp @@ -17,7 +17,7 @@ #include "Core/HW/DSP.h" #include "Core/HW/Memmap.h" #include "Core/PowerPC/Jit64/JitRegCache.h" -#include "Core/PowerPC/JitCommon/Jit_Util.h" +#include "Core/PowerPC/Jit64Common/Jit64Util.h" #include "Core/PowerPC/JitInterface.h" #include "Core/PowerPC/PowerPC.h" diff --git a/Source/Core/Core/PowerPC/Jit64/Jit_LoadStoreFloating.cpp b/Source/Core/Core/PowerPC/Jit64/Jit_LoadStoreFloating.cpp index 2b334d2680..2c6d29003f 100644 --- a/Source/Core/Core/PowerPC/Jit64/Jit_LoadStoreFloating.cpp +++ b/Source/Core/Core/PowerPC/Jit64/Jit_LoadStoreFloating.cpp @@ -4,11 +4,10 @@ #include "Core/PowerPC/Jit64/Jit.h" #include "Common/BitSet.h" -#include "Common/CPUDetect.h" #include "Common/CommonTypes.h" #include "Common/x64Emitter.h" #include "Core/PowerPC/Jit64/JitRegCache.h" -#include "Core/PowerPC/JitCommon/Jit_Util.h" +#include "Core/PowerPC/Jit64Common/Jit64Util.h" using namespace Gen; diff --git a/Source/Core/Core/PowerPC/Jit64/Jit_LoadStorePaired.cpp b/Source/Core/Core/PowerPC/Jit64/Jit_LoadStorePaired.cpp index 374da15668..c3f18e8a85 100644 --- a/Source/Core/Core/PowerPC/Jit64/Jit_LoadStorePaired.cpp +++ b/Source/Core/Core/PowerPC/Jit64/Jit_LoadStorePaired.cpp @@ -6,13 +6,12 @@ // Should give a very noticeable speed boost to paired single heavy code. #include "Core/PowerPC/Jit64/Jit.h" -#include "Common/BitSet.h" -#include "Common/CPUDetect.h" + #include "Common/CommonTypes.h" #include "Common/x64Emitter.h" #include "Core/PowerPC/Jit64/JitRegCache.h" +#include "Core/PowerPC/Jit64Common/Jit64Util.h" #include "Core/PowerPC/JitCommon/JitAsmCommon.h" -#include "Core/PowerPC/JitCommon/Jit_Util.h" #include "Core/PowerPC/PowerPC.h" using namespace Gen; diff --git a/Source/Core/Core/PowerPC/Jit64/Jit_SystemRegisters.cpp b/Source/Core/Core/PowerPC/Jit64/Jit_SystemRegisters.cpp index abc93dbca4..2f262540bd 100644 --- a/Source/Core/Core/PowerPC/Jit64/Jit_SystemRegisters.cpp +++ b/Source/Core/Core/PowerPC/Jit64/Jit_SystemRegisters.cpp @@ -9,7 +9,7 @@ #include "Core/CoreTiming.h" #include "Core/HW/ProcessorInterface.h" #include "Core/PowerPC/Jit64/JitRegCache.h" -#include "Core/PowerPC/JitCommon/Jit_Util.h" +#include "Core/PowerPC/Jit64Common/Jit64Util.h" #include "Core/PowerPC/PowerPC.h" using namespace Gen; diff --git a/Source/Core/Core/PowerPC/Jit64Common/BlockCache.cpp b/Source/Core/Core/PowerPC/Jit64Common/BlockCache.cpp new file mode 100644 index 0000000000..c19f471d19 --- /dev/null +++ b/Source/Core/Core/PowerPC/Jit64Common/BlockCache.cpp @@ -0,0 +1,40 @@ +// Copyright 2016 Dolphin Emulator Project +// Licensed under GPLv2+ +// Refer to the license.txt file included. + +#include "Core/PowerPC/Jit64Common/BlockCache.h" + +#include "Common/CommonTypes.h" +#include "Common/x64Emitter.h" +#include "Core/PowerPC/JitCommon/JitBase.h" + +void JitBlockCache::WriteLinkBlock(const JitBlock::LinkData& source, const JitBlock* dest) +{ + u8* location = source.exitPtrs; + const u8* address = dest ? dest->checkedEntry : jit->GetAsmRoutines()->dispatcher; + Gen::XEmitter emit(location); + if (*location == 0xE8) + { + emit.CALL(address); + } + else + { + // If we're going to link with the next block, there is no need + // to emit JMP. So just NOP out the gap to the next block. + // Support up to 3 additional bytes because of alignment. + s64 offset = address - emit.GetCodePtr(); + if (offset > 0 && offset <= 5 + 3) + emit.NOP(offset); + else + emit.JMP(address, true); + } +} + +void JitBlockCache::WriteDestroyBlock(const JitBlock& block) +{ + // Only clear the entry points as we might still be within this block. + Gen::XEmitter emit((u8*)block.checkedEntry); + emit.INT3(); + Gen::XEmitter emit2((u8*)block.normalEntry); + emit2.INT3(); +} diff --git a/Source/Core/Core/PowerPC/Jit64Common/BlockCache.h b/Source/Core/Core/PowerPC/Jit64Common/BlockCache.h new file mode 100644 index 0000000000..3d3f884e26 --- /dev/null +++ b/Source/Core/Core/PowerPC/Jit64Common/BlockCache.h @@ -0,0 +1,14 @@ +// Copyright 2016 Dolphin Emulator Project +// Licensed under GPLv2+ +// Refer to the license.txt file included. + +#pragma once + +#include "Core/PowerPC/JitCommon/JitCache.h" + +class JitBlockCache : public JitBaseBlockCache +{ +private: + void WriteLinkBlock(const JitBlock::LinkData& source, const JitBlock* dest) override; + void WriteDestroyBlock(const JitBlock& block) override; +}; diff --git a/Source/Core/Core/PowerPC/Jit64Common/Jit64AsmCommon.cpp b/Source/Core/Core/PowerPC/Jit64Common/Jit64AsmCommon.cpp index 2ac5bfb750..e1ab7425ab 100644 --- a/Source/Core/Core/PowerPC/Jit64Common/Jit64AsmCommon.cpp +++ b/Source/Core/Core/PowerPC/Jit64Common/Jit64AsmCommon.cpp @@ -11,8 +11,8 @@ #include "Common/x64Emitter.h" #include "Core/HW/GPFifo.h" #include "Core/PowerPC/Gekko.h" -#include "Core/PowerPC/JitCommon/JitBase.h" -#include "Core/PowerPC/JitCommon/Jit_Util.h" +#include "Core/PowerPC/Jit64Common/Jit64Base.h" +#include "Core/PowerPC/Jit64Common/Jit64Util.h" #include "Core/PowerPC/PowerPC.h" #define QUANTIZED_REGS_TO_SAVE \ diff --git a/Source/Core/Core/PowerPC/Jit64Common/Jit64AsmCommon.h b/Source/Core/Core/PowerPC/Jit64Common/Jit64AsmCommon.h index fc9f1d8bea..e4cdad983a 100644 --- a/Source/Core/Core/PowerPC/Jit64Common/Jit64AsmCommon.h +++ b/Source/Core/Core/PowerPC/Jit64Common/Jit64AsmCommon.h @@ -4,8 +4,9 @@ #pragma once +#include "Common/CommonTypes.h" +#include "Core/PowerPC/Jit64Common/Jit64Util.h" #include "Core/PowerPC/JitCommon/JitAsmCommon.h" -#include "Core/PowerPC/JitCommon/Jit_Util.h" enum EQuantizeType : u32; diff --git a/Source/Core/Core/PowerPC/Jit64Common/Jit64Base.cpp b/Source/Core/Core/PowerPC/Jit64Common/Jit64Base.cpp new file mode 100644 index 0000000000..f4b0fc86b9 --- /dev/null +++ b/Source/Core/Core/PowerPC/Jit64Common/Jit64Base.cpp @@ -0,0 +1,164 @@ +// Copyright 2016 Dolphin Emulator Project +// Licensed under GPLv2+ +// Refer to the license.txt file included. + +#include "Core/PowerPC/Jit64Common/Jit64Base.h" + +#include +#include +#include + +#include "Common/Assert.h" +#include "Common/CommonFuncs.h" +#include "Common/CommonTypes.h" +#include "Common/GekkoDisassembler.h" +#include "Common/Logging/Log.h" +#include "Common/MsgHandler.h" +#include "Common/StringUtil.h" +#include "Common/x64Reg.h" +#include "Core/HW/Memmap.h" +#include "Core/MachineContext.h" +#include "Core/PowerPC/PPCAnalyst.h" + +// This generates some fairly heavy trampolines, but it doesn't really hurt. +// Only instructions that access I/O will get these, and there won't be that +// many of them in a typical program/game. +bool Jitx86Base::HandleFault(uintptr_t access_address, SContext* ctx) +{ + // TODO: do we properly handle off-the-end? + if (access_address >= (uintptr_t)Memory::physical_base && + access_address < (uintptr_t)Memory::physical_base + 0x100010000) + return BackPatch((u32)(access_address - (uintptr_t)Memory::physical_base), ctx); + if (access_address >= (uintptr_t)Memory::logical_base && + access_address < (uintptr_t)Memory::logical_base + 0x100010000) + return BackPatch((u32)(access_address - (uintptr_t)Memory::logical_base), ctx); + + return false; +} + +bool Jitx86Base::BackPatch(u32 emAddress, SContext* ctx) +{ + u8* codePtr = (u8*)ctx->CTX_PC; + + if (!IsInSpace(codePtr)) + return false; // this will become a regular crash real soon after this + + auto it = backPatchInfo.find(codePtr); + if (it == backPatchInfo.end()) + { + PanicAlert("BackPatch: no register use entry for address %p", codePtr); + return false; + } + + TrampolineInfo& info = it->second; + + u8* exceptionHandler = nullptr; + if (jit->jo.memcheck) + { + auto it2 = exceptionHandlerAtLoc.find(codePtr); + if (it2 != exceptionHandlerAtLoc.end()) + exceptionHandler = it2->second; + } + + // In the trampoline code, we jump back into the block at the beginning + // of the next instruction. The next instruction comes immediately + // after the backpatched operation, or BACKPATCH_SIZE bytes after the start + // of the backpatched operation, whichever comes last. (The JIT inserts NOPs + // into the original code if necessary to ensure there is enough space + // to insert the backpatch jump.) + + jit->js.generatingTrampoline = true; + jit->js.trampolineExceptionHandler = exceptionHandler; + + // Generate the trampoline. + const u8* trampoline = trampolines.GenerateTrampoline(info); + jit->js.generatingTrampoline = false; + jit->js.trampolineExceptionHandler = nullptr; + + u8* start = info.start; + + // Patch the original memory operation. + XEmitter emitter(start); + emitter.JMP(trampoline, true); + // NOPs become dead code + const u8* end = info.start + info.len; + for (const u8* i = emitter.GetCodePtr(); i < end; ++i) + emitter.INT3(); + + // Rewind time to just before the start of the write block. If we swapped memory + // before faulting (eg: the store+swap was not an atomic op like MOVBE), let's + // swap it back so that the swap can happen again (this double swap isn't ideal but + // only happens the first time we fault). + if (info.nonAtomicSwapStoreSrc != Gen::INVALID_REG) + { + u64* ptr = ContextRN(ctx, info.nonAtomicSwapStoreSrc); + switch (info.accessSize << 3) + { + case 8: + // No need to swap a byte + break; + case 16: + *ptr = Common::swap16(static_cast(*ptr)); + break; + case 32: + *ptr = Common::swap32(static_cast(*ptr)); + break; + case 64: + *ptr = Common::swap64(static_cast(*ptr)); + break; + default: + _dbg_assert_(DYNA_REC, 0); + break; + } + } + + // This is special code to undo the LEA in SafeLoadToReg if it clobbered the address + // register in the case where reg_value shared the same location as opAddress. + if (info.offsetAddedToAddress) + { + u64* ptr = ContextRN(ctx, info.op_arg.GetSimpleReg()); + *ptr -= static_cast(info.offset); + } + + ctx->CTX_PC = reinterpret_cast(trampoline); + + return true; +} + +void LogGeneratedX86(int size, PPCAnalyst::CodeBuffer* code_buffer, const u8* normalEntry, + JitBlock* b) +{ + for (int i = 0; i < size; i++) + { + const PPCAnalyst::CodeOp& op = code_buffer->codebuffer[i]; + std::string temp = StringFromFormat( + "%08x %s", op.address, GekkoDisassembler::Disassemble(op.inst.hex, op.address).c_str()); + DEBUG_LOG(DYNA_REC, "IR_X86 PPC: %s\n", temp.c_str()); + } + + disassembler x64disasm; + x64disasm.set_syntax_intel(); + + u64 disasmPtr = (u64)normalEntry; + const u8* end = normalEntry + b->codeSize; + + while ((u8*)disasmPtr < end) + { + char sptr[1000] = ""; + disasmPtr += x64disasm.disasm64(disasmPtr, disasmPtr, (u8*)disasmPtr, sptr); + DEBUG_LOG(DYNA_REC, "IR_X86 x86: %s", sptr); + } + + if (b->codeSize <= 250) + { + std::stringstream ss; + ss << std::hex; + for (u8 i = 0; i <= b->codeSize; i++) + { + ss.width(2); + ss.fill('0'); + ss << (u32) * (normalEntry + i); + } + DEBUG_LOG(DYNA_REC, "IR_X86 bin: %s\n\n\n", ss.str().c_str()); + } +} diff --git a/Source/Core/Core/PowerPC/Jit64Common/Jit64Base.h b/Source/Core/Core/PowerPC/Jit64Common/Jit64Base.h new file mode 100644 index 0000000000..8a21f47cdb --- /dev/null +++ b/Source/Core/Core/PowerPC/Jit64Common/Jit64Base.h @@ -0,0 +1,48 @@ +// Copyright 2016 Dolphin Emulator Project +// Licensed under GPLv2+ +// Refer to the license.txt file included. + +#pragma once + +#include + +#include "Common/CommonTypes.h" +#include "Common/x64Reg.h" +#include "Core/PowerPC/Jit64Common/BlockCache.h" +#include "Core/PowerPC/Jit64Common/Jit64AsmCommon.h" +#include "Core/PowerPC/Jit64Common/TrampolineCache.h" +#include "Core/PowerPC/JitCommon/JitBase.h" + +namespace PPCAnalyst +{ +class CodeBuffer; +} + +// The following register assignments are common to Jit64 and Jit64IL: +// RSCRATCH and RSCRATCH2 are always scratch registers and can be used without +// limitation. +#define RSCRATCH RAX +#define RSCRATCH2 RDX +// RSCRATCH_EXTRA may be in the allocation order, so it has to be flushed +// before use. +#define RSCRATCH_EXTRA RCX +// RMEM points to the start of emulated memory. +#define RMEM RBX +// RPPCSTATE points to ppcState + 0x80. It's offset because we want to be able +// to address as much as possible in a one-byte offset form. +#define RPPCSTATE RBP + +class Jitx86Base : public JitBase, public QuantizedMemoryRoutines +{ +protected: + bool BackPatch(u32 emAddress, SContext* ctx); + JitBlockCache blocks; + TrampolineCache trampolines; + +public: + JitBlockCache* GetBlockCache() override { return &blocks; } + bool HandleFault(uintptr_t access_address, SContext* ctx) override; +}; + +void LogGeneratedX86(int size, PPCAnalyst::CodeBuffer* code_buffer, const u8* normalEntry, + JitBlock* b); diff --git a/Source/Core/Core/PowerPC/Jit64Common/Jit64Util.cpp b/Source/Core/Core/PowerPC/Jit64Common/Jit64Util.cpp new file mode 100644 index 0000000000..9db6e5b919 --- /dev/null +++ b/Source/Core/Core/PowerPC/Jit64Common/Jit64Util.cpp @@ -0,0 +1,1114 @@ +// Copyright 2008 Dolphin Emulator Project +// Licensed under GPLv2+ +// Refer to the license.txt file included. + +#include "Core/PowerPC/Jit64Common/Jit64Util.h" +#include "Common/BitSet.h" +#include "Common/CommonTypes.h" +#include "Common/Intrinsics.h" +#include "Common/MathUtil.h" +#include "Common/x64ABI.h" +#include "Common/x64Emitter.h" +#include "Core/HW/MMIO.h" +#include "Core/HW/Memmap.h" +#include "Core/PowerPC/Jit64Common/Jit64Base.h" +#include "Core/PowerPC/Jit64Common/TrampolineCache.h" +#include "Core/PowerPC/PowerPC.h" + +using namespace Gen; + +void EmuCodeBlock::MemoryExceptionCheck() +{ + // TODO: We really should untangle the trampolines, exception handlers and + // memory checks. + + // If we are currently generating a trampoline for a failed fastmem + // load/store, the trampoline generator will have stashed the exception + // handler (that we previously generated after the fastmem instruction) in + // trampolineExceptionHandler. + if (jit->js.generatingTrampoline) + { + if (jit->js.trampolineExceptionHandler) + { + TEST(32, PPCSTATE(Exceptions), Gen::Imm32(EXCEPTION_DSI)); + J_CC(CC_NZ, jit->js.trampolineExceptionHandler); + } + return; + } + + // If memcheck (ie: MMU) mode is enabled and we haven't generated an + // exception handler for this instruction yet, we will generate an + // exception check. + if (jit->jo.memcheck && !jit->js.fastmemLoadStore && !jit->js.fixupExceptionHandler) + { + TEST(32, PPCSTATE(Exceptions), Gen::Imm32(EXCEPTION_DSI)); + jit->js.exceptionHandler = J_CC(Gen::CC_NZ, true); + jit->js.fixupExceptionHandler = true; + } +} + +void EmuCodeBlock::UnsafeLoadRegToReg(X64Reg reg_addr, X64Reg reg_value, int accessSize, s32 offset, + bool signExtend) +{ + OpArg src = MComplex(RMEM, reg_addr, SCALE_1, offset); + LoadAndSwap(accessSize, reg_value, src, signExtend); +} + +void EmuCodeBlock::UnsafeLoadRegToRegNoSwap(X64Reg reg_addr, X64Reg reg_value, int accessSize, + s32 offset, bool signExtend) +{ + if (signExtend) + MOVSX(32, accessSize, reg_value, MComplex(RMEM, reg_addr, SCALE_1, offset)); + else + MOVZX(32, accessSize, reg_value, MComplex(RMEM, reg_addr, SCALE_1, offset)); +} + +bool EmuCodeBlock::UnsafeLoadToReg(X64Reg reg_value, OpArg opAddress, int accessSize, s32 offset, + bool signExtend, MovInfo* info) +{ + bool offsetAddedToAddress = false; + OpArg memOperand; + if (opAddress.IsSimpleReg()) + { + // Deal with potential wraparound. (This is just a heuristic, and it would + // be more correct to actually mirror the first page at the end, but the + // only case where it probably actually matters is JitIL turning adds into + // offsets with the wrong sign, so whatever. Since the original code + // *could* try to wrap an address around, however, this is the correct + // place to address the issue.) + if ((u32)offset >= 0x1000) + { + // This method can potentially clobber the address if it shares a register + // with the load target. In this case we can just subtract offset from the + // register (see Jit64Base for this implementation). + offsetAddedToAddress = (reg_value == opAddress.GetSimpleReg()); + + LEA(32, reg_value, MDisp(opAddress.GetSimpleReg(), offset)); + opAddress = R(reg_value); + offset = 0; + } + memOperand = MComplex(RMEM, opAddress.GetSimpleReg(), SCALE_1, offset); + } + else if (opAddress.IsImm()) + { + MOV(32, R(reg_value), Imm32((u32)(opAddress.Imm32() + offset))); + memOperand = MRegSum(RMEM, reg_value); + } + else + { + MOV(32, R(reg_value), opAddress); + memOperand = MComplex(RMEM, reg_value, SCALE_1, offset); + } + + LoadAndSwap(accessSize, reg_value, memOperand, signExtend, info); + return offsetAddedToAddress; +} + +// Visitor that generates code to read a MMIO value. +template +class MMIOReadCodeGenerator : public MMIO::ReadHandlingMethodVisitor +{ +public: + MMIOReadCodeGenerator(Gen::X64CodeBlock* code, BitSet32 registers_in_use, Gen::X64Reg dst_reg, + u32 address, bool sign_extend) + : m_code(code), m_registers_in_use(registers_in_use), m_dst_reg(dst_reg), m_address(address), + m_sign_extend(sign_extend) + { + } + + void VisitConstant(T value) override { LoadConstantToReg(8 * sizeof(T), value); } + void VisitDirect(const T* addr, u32 mask) override + { + LoadAddrMaskToReg(8 * sizeof(T), addr, mask); + } + void VisitComplex(const std::function* lambda) override + { + CallLambda(8 * sizeof(T), lambda); + } + +private: + // Generates code to load a constant to the destination register. In + // practice it would be better to avoid using a register for this, but it + // would require refactoring a lot of JIT code. + void LoadConstantToReg(int sbits, u32 value) + { + if (m_sign_extend) + { + u32 sign = !!(value & (1 << (sbits - 1))); + value |= sign * ((0xFFFFFFFF >> sbits) << sbits); + } + m_code->MOV(32, R(m_dst_reg), Gen::Imm32(value)); + } + + // Generate the proper MOV instruction depending on whether the read should + // be sign extended or zero extended. + void MoveOpArgToReg(int sbits, const Gen::OpArg& arg) + { + if (m_sign_extend) + m_code->MOVSX(32, sbits, m_dst_reg, arg); + else + m_code->MOVZX(32, sbits, m_dst_reg, arg); + } + + void LoadAddrMaskToReg(int sbits, const void* ptr, u32 mask) + { + m_code->MOV(64, R(RSCRATCH), ImmPtr(ptr)); + // If we do not need to mask, we can do the sign extend while loading + // from memory. If masking is required, we have to first zero extend, + // then mask, then sign extend if needed (1 instr vs. 2/3). + u32 all_ones = (1ULL << sbits) - 1; + if ((all_ones & mask) == all_ones) + { + MoveOpArgToReg(sbits, MatR(RSCRATCH)); + } + else + { + m_code->MOVZX(32, sbits, m_dst_reg, MatR(RSCRATCH)); + m_code->AND(32, R(m_dst_reg), Imm32(mask)); + if (m_sign_extend) + m_code->MOVSX(32, sbits, m_dst_reg, R(m_dst_reg)); + } + } + + void CallLambda(int sbits, const std::function* lambda) + { + m_code->ABI_PushRegistersAndAdjustStack(m_registers_in_use, 0); + m_code->ABI_CallLambdaC(lambda, m_address); + m_code->ABI_PopRegistersAndAdjustStack(m_registers_in_use, 0); + MoveOpArgToReg(sbits, R(ABI_RETURN)); + } + + Gen::X64CodeBlock* m_code; + BitSet32 m_registers_in_use; + Gen::X64Reg m_dst_reg; + u32 m_address; + bool m_sign_extend; +}; + +void EmuCodeBlock::MMIOLoadToReg(MMIO::Mapping* mmio, Gen::X64Reg reg_value, + BitSet32 registers_in_use, u32 address, int access_size, + bool sign_extend) +{ + switch (access_size) + { + case 8: + { + MMIOReadCodeGenerator gen(this, registers_in_use, reg_value, address, sign_extend); + mmio->GetHandlerForRead(address).Visit(gen); + break; + } + case 16: + { + MMIOReadCodeGenerator gen(this, registers_in_use, reg_value, address, sign_extend); + mmio->GetHandlerForRead(address).Visit(gen); + break; + } + case 32: + { + MMIOReadCodeGenerator gen(this, registers_in_use, reg_value, address, sign_extend); + mmio->GetHandlerForRead(address).Visit(gen); + break; + } + } +} + +FixupBranch EmuCodeBlock::CheckIfSafeAddress(const OpArg& reg_value, X64Reg reg_addr, + BitSet32 registers_in_use) +{ + registers_in_use[reg_addr] = true; + if (reg_value.IsSimpleReg()) + registers_in_use[reg_value.GetSimpleReg()] = true; + + // Get ourselves a free register; try to pick one that doesn't involve pushing, if we can. + X64Reg scratch = RSCRATCH; + if (!registers_in_use[RSCRATCH]) + scratch = RSCRATCH; + else if (!registers_in_use[RSCRATCH_EXTRA]) + scratch = RSCRATCH_EXTRA; + else + scratch = reg_addr; + + if (scratch == reg_addr) + PUSH(scratch); + else + MOV(32, R(scratch), R(reg_addr)); + + // Perform lookup to see if we can use fast path. + SHR(32, R(scratch), Imm8(PowerPC::BAT_INDEX_SHIFT)); + TEST(32, MScaled(scratch, SCALE_4, PtrOffset(&PowerPC::dbat_table[0])), Imm32(2)); + + if (scratch == reg_addr) + POP(scratch); + + return J_CC(CC_Z, farcode.Enabled()); +} + +void EmuCodeBlock::SafeLoadToReg(X64Reg reg_value, const Gen::OpArg& opAddress, int accessSize, + s32 offset, BitSet32 registersInUse, bool signExtend, int flags) +{ + bool slowmem = (flags & SAFE_LOADSTORE_FORCE_SLOWMEM) != 0 || jit->jo.alwaysUseMemFuncs; + + registersInUse[reg_value] = false; + if (jit->jo.fastmem && !(flags & SAFE_LOADSTORE_NO_FASTMEM) && !slowmem) + { + u8* backpatchStart = GetWritableCodePtr(); + MovInfo mov; + bool offsetAddedToAddress = + UnsafeLoadToReg(reg_value, opAddress, accessSize, offset, signExtend, &mov); + TrampolineInfo& info = backPatchInfo[mov.address]; + info.pc = jit->js.compilerPC; + info.nonAtomicSwapStoreSrc = mov.nonAtomicSwapStore ? mov.nonAtomicSwapStoreSrc : INVALID_REG; + info.start = backpatchStart; + info.read = true; + info.op_reg = reg_value; + info.op_arg = opAddress; + info.offsetAddedToAddress = offsetAddedToAddress; + info.accessSize = accessSize >> 3; + info.offset = offset; + info.registersInUse = registersInUse; + info.flags = flags; + info.signExtend = signExtend; + ptrdiff_t padding = BACKPATCH_SIZE - (GetCodePtr() - backpatchStart); + if (padding > 0) + { + NOP(padding); + } + info.len = static_cast(GetCodePtr() - info.start); + + jit->js.fastmemLoadStore = mov.address; + return; + } + + if (opAddress.IsImm()) + { + u32 address = opAddress.Imm32() + offset; + SafeLoadToRegImmediate(reg_value, address, accessSize, registersInUse, signExtend); + return; + } + + _assert_msg_(DYNA_REC, opAddress.IsSimpleReg(), + "Incorrect use of SafeLoadToReg (address isn't register or immediate)"); + X64Reg reg_addr = opAddress.GetSimpleReg(); + if (offset) + { + reg_addr = RSCRATCH; + LEA(32, RSCRATCH, MDisp(opAddress.GetSimpleReg(), offset)); + } + + FixupBranch exit; + bool dr_set = (flags & SAFE_LOADSTORE_DR_ON) || UReg_MSR(MSR).DR; + bool fast_check_address = !slowmem && dr_set; + if (fast_check_address) + { + FixupBranch slow = CheckIfSafeAddress(R(reg_value), reg_addr, registersInUse); + UnsafeLoadToReg(reg_value, R(reg_addr), accessSize, 0, signExtend); + if (farcode.Enabled()) + SwitchToFarCode(); + else + exit = J(true); + SetJumpTarget(slow); + } + size_t rsp_alignment = (flags & SAFE_LOADSTORE_NO_PROLOG) ? 8 : 0; + ABI_PushRegistersAndAdjustStack(registersInUse, rsp_alignment); + switch (accessSize) + { + case 64: + ABI_CallFunctionR(PowerPC::Read_U64, reg_addr); + break; + case 32: + ABI_CallFunctionR(PowerPC::Read_U32, reg_addr); + break; + case 16: + ABI_CallFunctionR(PowerPC::Read_U16_ZX, reg_addr); + break; + case 8: + ABI_CallFunctionR(PowerPC::Read_U8_ZX, reg_addr); + break; + } + ABI_PopRegistersAndAdjustStack(registersInUse, rsp_alignment); + + MemoryExceptionCheck(); + if (signExtend && accessSize < 32) + { + // Need to sign extend values coming from the Read_U* functions. + MOVSX(32, accessSize, reg_value, R(ABI_RETURN)); + } + else if (reg_value != ABI_RETURN) + { + MOVZX(64, accessSize, reg_value, R(ABI_RETURN)); + } + + if (fast_check_address) + { + if (farcode.Enabled()) + { + exit = J(true); + SwitchToNearCode(); + } + SetJumpTarget(exit); + } +} + +void EmuCodeBlock::SafeLoadToRegImmediate(X64Reg reg_value, u32 address, int accessSize, + BitSet32 registersInUse, bool signExtend) +{ + // If the address is known to be RAM, just load it directly. + if (PowerPC::IsOptimizableRAMAddress(address)) + { + UnsafeLoadToReg(reg_value, Imm32(address), accessSize, 0, signExtend); + return; + } + + // If the address maps to an MMIO register, inline MMIO read code. + u32 mmioAddress = PowerPC::IsOptimizableMMIOAccess(address, accessSize); + if (accessSize != 64 && mmioAddress) + { + MMIOLoadToReg(Memory::mmio_mapping.get(), reg_value, registersInUse, mmioAddress, accessSize, + signExtend); + return; + } + + // Fall back to general-case code. + ABI_PushRegistersAndAdjustStack(registersInUse, 0); + switch (accessSize) + { + case 64: + ABI_CallFunctionC(PowerPC::Read_U64, address); + break; + case 32: + ABI_CallFunctionC(PowerPC::Read_U32, address); + break; + case 16: + ABI_CallFunctionC(PowerPC::Read_U16_ZX, address); + break; + case 8: + ABI_CallFunctionC(PowerPC::Read_U8_ZX, address); + break; + } + ABI_PopRegistersAndAdjustStack(registersInUse, 0); + + MemoryExceptionCheck(); + if (signExtend && accessSize < 32) + { + // Need to sign extend values coming from the Read_U* functions. + MOVSX(32, accessSize, reg_value, R(ABI_RETURN)); + } + else if (reg_value != ABI_RETURN) + { + MOVZX(64, accessSize, reg_value, R(ABI_RETURN)); + } +} + +static OpArg SwapImmediate(int accessSize, const OpArg& reg_value) +{ + if (accessSize == 32) + return Imm32(Common::swap32(reg_value.Imm32())); + else if (accessSize == 16) + return Imm16(Common::swap16(reg_value.Imm16())); + else + return Imm8(reg_value.Imm8()); +} + +void EmuCodeBlock::UnsafeWriteRegToReg(OpArg reg_value, X64Reg reg_addr, int accessSize, s32 offset, + bool swap, MovInfo* info) +{ + if (info) + { + info->address = GetWritableCodePtr(); + info->nonAtomicSwapStore = false; + } + + OpArg dest = MComplex(RMEM, reg_addr, SCALE_1, offset); + if (reg_value.IsImm()) + { + if (swap) + reg_value = SwapImmediate(accessSize, reg_value); + MOV(accessSize, dest, reg_value); + } + else if (swap) + { + SwapAndStore(accessSize, dest, reg_value.GetSimpleReg(), info); + } + else + { + MOV(accessSize, dest, reg_value); + } +} + +static OpArg FixImmediate(int accessSize, OpArg arg) +{ + if (arg.IsImm()) + { + arg = accessSize == 8 ? arg.AsImm8() : accessSize == 16 ? arg.AsImm16() : arg.AsImm32(); + } + return arg; +} + +void EmuCodeBlock::UnsafeWriteGatherPipe(int accessSize) +{ + // No need to protect these, they don't touch any state + // question - should we inline them instead? Pro: Lose a CALL Con: Code bloat + switch (accessSize) + { + case 8: + CALL(jit->GetAsmRoutines()->fifoDirectWrite8); + break; + case 16: + CALL(jit->GetAsmRoutines()->fifoDirectWrite16); + break; + case 32: + CALL(jit->GetAsmRoutines()->fifoDirectWrite32); + break; + case 64: + CALL(jit->GetAsmRoutines()->fifoDirectWrite64); + break; + } + jit->js.fifoBytesSinceCheck += accessSize >> 3; +} + +bool EmuCodeBlock::WriteToConstAddress(int accessSize, OpArg arg, u32 address, + BitSet32 registersInUse) +{ + arg = FixImmediate(accessSize, arg); + + // If we already know the address through constant folding, we can do some + // fun tricks... + if (jit->jo.optimizeGatherPipe && PowerPC::IsOptimizableGatherPipeWrite(address)) + { + if (!arg.IsSimpleReg(RSCRATCH)) + MOV(accessSize, R(RSCRATCH), arg); + + UnsafeWriteGatherPipe(accessSize); + return false; + } + else if (PowerPC::IsOptimizableRAMAddress(address)) + { + WriteToConstRamAddress(accessSize, arg, address); + return false; + } + else + { + // Helps external systems know which instruction triggered the write + MOV(32, PPCSTATE(pc), Imm32(jit->js.compilerPC)); + + ABI_PushRegistersAndAdjustStack(registersInUse, 0); + switch (accessSize) + { + case 64: + ABI_CallFunctionAC(64, PowerPC::Write_U64, arg, address); + break; + case 32: + ABI_CallFunctionAC(32, PowerPC::Write_U32, arg, address); + break; + case 16: + ABI_CallFunctionAC(16, PowerPC::Write_U16, arg, address); + break; + case 8: + ABI_CallFunctionAC(8, PowerPC::Write_U8, arg, address); + break; + } + ABI_PopRegistersAndAdjustStack(registersInUse, 0); + return true; + } +} + +void EmuCodeBlock::SafeWriteRegToReg(OpArg reg_value, X64Reg reg_addr, int accessSize, s32 offset, + BitSet32 registersInUse, int flags) +{ + bool swap = !(flags & SAFE_LOADSTORE_NO_SWAP); + bool slowmem = (flags & SAFE_LOADSTORE_FORCE_SLOWMEM) != 0 || jit->jo.alwaysUseMemFuncs; + + // set the correct immediate format + reg_value = FixImmediate(accessSize, reg_value); + + if (jit->jo.fastmem && !(flags & SAFE_LOADSTORE_NO_FASTMEM) && !slowmem) + { + u8* backpatchStart = GetWritableCodePtr(); + MovInfo mov; + UnsafeWriteRegToReg(reg_value, reg_addr, accessSize, offset, swap, &mov); + TrampolineInfo& info = backPatchInfo[mov.address]; + info.pc = jit->js.compilerPC; + info.nonAtomicSwapStoreSrc = mov.nonAtomicSwapStore ? mov.nonAtomicSwapStoreSrc : INVALID_REG; + info.start = backpatchStart; + info.read = false; + info.op_arg = reg_value; + info.op_reg = reg_addr; + info.offsetAddedToAddress = false; + info.accessSize = accessSize >> 3; + info.offset = offset; + info.registersInUse = registersInUse; + info.flags = flags; + ptrdiff_t padding = BACKPATCH_SIZE - (GetCodePtr() - backpatchStart); + if (padding > 0) + { + NOP(padding); + } + info.len = static_cast(GetCodePtr() - info.start); + + jit->js.fastmemLoadStore = mov.address; + + return; + } + + if (offset) + { + if (flags & SAFE_LOADSTORE_CLOBBER_RSCRATCH_INSTEAD_OF_ADDR) + { + LEA(32, RSCRATCH, MDisp(reg_addr, (u32)offset)); + reg_addr = RSCRATCH; + } + else + { + ADD(32, R(reg_addr), Imm32((u32)offset)); + } + } + + FixupBranch exit; + bool dr_set = (flags & SAFE_LOADSTORE_DR_ON) || UReg_MSR(MSR).DR; + bool fast_check_address = !slowmem && dr_set; + if (fast_check_address) + { + FixupBranch slow = CheckIfSafeAddress(reg_value, reg_addr, registersInUse); + UnsafeWriteRegToReg(reg_value, reg_addr, accessSize, 0, swap); + if (farcode.Enabled()) + SwitchToFarCode(); + else + exit = J(true); + SetJumpTarget(slow); + } + + // PC is used by memory watchpoints (if enabled) or to print accurate PC locations in debug logs + MOV(32, PPCSTATE(pc), Imm32(jit->js.compilerPC)); + + size_t rsp_alignment = (flags & SAFE_LOADSTORE_NO_PROLOG) ? 8 : 0; + ABI_PushRegistersAndAdjustStack(registersInUse, rsp_alignment); + + // If the input is an immediate, we need to put it in a register. + X64Reg reg; + if (reg_value.IsImm()) + { + reg = reg_addr == ABI_PARAM1 ? RSCRATCH : ABI_PARAM1; + MOV(accessSize, R(reg), reg_value); + } + else + { + reg = reg_value.GetSimpleReg(); + } + + switch (accessSize) + { + case 64: + ABI_CallFunctionRR(swap ? PowerPC::Write_U64 : PowerPC::Write_U64_Swap, reg, reg_addr); + break; + case 32: + ABI_CallFunctionRR(swap ? PowerPC::Write_U32 : PowerPC::Write_U32_Swap, reg, reg_addr); + break; + case 16: + ABI_CallFunctionRR(swap ? PowerPC::Write_U16 : PowerPC::Write_U16_Swap, reg, reg_addr); + break; + case 8: + ABI_CallFunctionRR(PowerPC::Write_U8, reg, reg_addr); + break; + } + ABI_PopRegistersAndAdjustStack(registersInUse, rsp_alignment); + + MemoryExceptionCheck(); + + if (fast_check_address) + { + if (farcode.Enabled()) + { + exit = J(true); + SwitchToNearCode(); + } + SetJumpTarget(exit); + } +} + +void EmuCodeBlock::WriteToConstRamAddress(int accessSize, OpArg arg, u32 address, bool swap) +{ + X64Reg reg; + if (arg.IsImm()) + { + arg = SwapImmediate(accessSize, arg); + MOV(32, R(RSCRATCH), Imm32(address)); + MOV(accessSize, MRegSum(RMEM, RSCRATCH), arg); + return; + } + + if (!arg.IsSimpleReg() || (!cpu_info.bMOVBE && swap && arg.GetSimpleReg() != RSCRATCH)) + { + MOV(accessSize, R(RSCRATCH), arg); + reg = RSCRATCH; + } + else + { + reg = arg.GetSimpleReg(); + } + + MOV(32, R(RSCRATCH2), Imm32(address)); + if (swap) + SwapAndStore(accessSize, MRegSum(RMEM, RSCRATCH2), reg); + else + MOV(accessSize, MRegSum(RMEM, RSCRATCH2), R(reg)); +} + +void EmuCodeBlock::ForceSinglePrecision(X64Reg output, const OpArg& input, bool packed, + bool duplicate) +{ + // Most games don't need these. Zelda requires it though - some platforms get stuck without them. + if (jit->jo.accurateSinglePrecision) + { + if (packed) + { + CVTPD2PS(output, input); + CVTPS2PD(output, R(output)); + } + else + { + CVTSD2SS(output, input); + CVTSS2SD(output, R(output)); + if (duplicate) + MOVDDUP(output, R(output)); + } + } + else if (!input.IsSimpleReg(output)) + { + if (duplicate) + MOVDDUP(output, input); + else + MOVAPD(output, input); + } +} + +// Abstract between AVX and SSE: automatically handle 3-operand instructions +void EmuCodeBlock::avx_op(void (XEmitter::*avxOp)(X64Reg, X64Reg, const OpArg&), + void (XEmitter::*sseOp)(X64Reg, const OpArg&), X64Reg regOp, + const OpArg& arg1, const OpArg& arg2, bool packed, bool reversible) +{ + if (arg1.IsSimpleReg(regOp)) + { + (this->*sseOp)(regOp, arg2); + } + else if (arg1.IsSimpleReg() && cpu_info.bAVX) + { + (this->*avxOp)(regOp, arg1.GetSimpleReg(), arg2); + } + else if (arg2.IsSimpleReg(regOp)) + { + if (reversible) + { + (this->*sseOp)(regOp, arg1); + } + else + { + // The ugly case: regOp == arg2 without AVX, or with arg1 == memory + if (!arg1.IsSimpleReg(XMM0)) + MOVAPD(XMM0, arg1); + if (cpu_info.bAVX) + { + (this->*avxOp)(regOp, XMM0, arg2); + } + else + { + (this->*sseOp)(XMM0, arg2); + if (packed) + MOVAPD(regOp, R(XMM0)); + else + MOVSD(regOp, R(XMM0)); + } + } + } + else + { + if (packed) + MOVAPD(regOp, arg1); + else + MOVSD(regOp, arg1); + (this->*sseOp)(regOp, arg1 == arg2 ? R(regOp) : arg2); + } +} + +// Abstract between AVX and SSE: automatically handle 3-operand instructions +void EmuCodeBlock::avx_op(void (XEmitter::*avxOp)(X64Reg, X64Reg, const OpArg&, u8), + void (XEmitter::*sseOp)(X64Reg, const OpArg&, u8), X64Reg regOp, + const OpArg& arg1, const OpArg& arg2, u8 imm) +{ + if (arg1.IsSimpleReg(regOp)) + { + (this->*sseOp)(regOp, arg2, imm); + } + else if (arg1.IsSimpleReg() && cpu_info.bAVX) + { + (this->*avxOp)(regOp, arg1.GetSimpleReg(), arg2, imm); + } + else if (arg2.IsSimpleReg(regOp)) + { + // The ugly case: regOp == arg2 without AVX, or with arg1 == memory + if (!arg1.IsSimpleReg(XMM0)) + MOVAPD(XMM0, arg1); + if (cpu_info.bAVX) + { + (this->*avxOp)(regOp, XMM0, arg2, imm); + } + else + { + (this->*sseOp)(XMM0, arg2, imm); + MOVAPD(regOp, R(XMM0)); + } + } + else + { + MOVAPD(regOp, arg1); + (this->*sseOp)(regOp, arg1 == arg2 ? R(regOp) : arg2, imm); + } +} + +alignas(16) static const u64 psMantissaTruncate[2] = {0xFFFFFFFFF8000000ULL, 0xFFFFFFFFF8000000ULL}; +alignas(16) static const u64 psRoundBit[2] = {0x8000000, 0x8000000}; + +// Emulate the odd truncation/rounding that the PowerPC does on the RHS operand before +// a single precision multiply. To be precise, it drops the low 28 bits of the mantissa, +// rounding to nearest as it does. +// It needs a temp, so let the caller pass that in. +void EmuCodeBlock::Force25BitPrecision(X64Reg output, const OpArg& input, X64Reg tmp) +{ + if (jit->jo.accurateSinglePrecision) + { + // mantissa = (mantissa & ~0xFFFFFFF) + ((mantissa & (1ULL << 27)) << 1); + if (input.IsSimpleReg() && cpu_info.bAVX) + { + VPAND(tmp, input.GetSimpleReg(), M(psRoundBit)); + VPAND(output, input.GetSimpleReg(), M(psMantissaTruncate)); + PADDQ(output, R(tmp)); + } + else + { + if (!input.IsSimpleReg(output)) + MOVAPD(output, input); + avx_op(&XEmitter::VPAND, &XEmitter::PAND, tmp, R(output), M(psRoundBit), true, true); + PAND(output, M(psMantissaTruncate)); + PADDQ(output, R(tmp)); + } + } + else if (!input.IsSimpleReg(output)) + { + MOVAPD(output, input); + } +} + +alignas(16) static u32 temp32; +alignas(16) static u64 temp64; + +// Since the following float conversion functions are used in non-arithmetic PPC float instructions, +// they must convert floats bitexact and never flush denormals to zero or turn SNaNs into QNaNs. +// This means we can't use CVTSS2SD/CVTSD2SS :( +// The x87 FPU doesn't even support flush-to-zero so we can use FLD+FSTP even on denormals. +// If the number is a NaN, make sure to set the QNaN bit back to its original value. + +// Another problem is that officially, converting doubles to single format results in undefined +// behavior. +// Relying on undefined behavior is a bug so no software should ever do this. +// In case it does happen, phire's more accurate implementation of ConvertDoubleToSingle() is +// reproduced below. + +//#define MORE_ACCURATE_DOUBLETOSINGLE +#ifdef MORE_ACCURATE_DOUBLETOSINGLE + +alignas(16) static const __m128i double_exponent = _mm_set_epi64x(0, 0x7ff0000000000000); +alignas(16) static const __m128i double_fraction = _mm_set_epi64x(0, 0x000fffffffffffff); +alignas(16) static const __m128i double_sign_bit = _mm_set_epi64x(0, 0x8000000000000000); +alignas(16) static const __m128i double_explicit_top_bit = _mm_set_epi64x(0, 0x0010000000000000); +alignas(16) static const __m128i double_top_two_bits = _mm_set_epi64x(0, 0xc000000000000000); +alignas(16) static const __m128i double_bottom_bits = _mm_set_epi64x(0, 0x07ffffffe0000000); + +// This is the same algorithm used in the interpreter (and actual hardware) +// The documentation states that the conversion of a double with an outside the +// valid range for a single (or a single denormal) is undefined. +// But testing on actual hardware shows it always picks bits 0..1 and 5..34 +// unless the exponent is in the range of 874 to 896. +void EmuCodeBlock::ConvertDoubleToSingle(X64Reg dst, X64Reg src) +{ + MOVSD(XMM1, R(src)); + + // Grab Exponent + PAND(XMM1, M(&double_exponent)); + PSRLQ(XMM1, 52); + MOVD_xmm(R(RSCRATCH), XMM1); + + // Check if the double is in the range of valid single subnormal + CMP(16, R(RSCRATCH), Imm16(896)); + FixupBranch NoDenormalize = J_CC(CC_G); + CMP(16, R(RSCRATCH), Imm16(874)); + FixupBranch NoDenormalize2 = J_CC(CC_L); + + // Denormalise + + // shift = (905 - Exponent) plus the 21 bit double to single shift + MOV(16, R(RSCRATCH), Imm16(905 + 21)); + MOVD_xmm(XMM0, R(RSCRATCH)); + PSUBQ(XMM0, R(XMM1)); + + // xmm1 = fraction | 0x0010000000000000 + MOVSD(XMM1, R(src)); + PAND(XMM1, M(&double_fraction)); + POR(XMM1, M(&double_explicit_top_bit)); + + // fraction >> shift + PSRLQ(XMM1, R(XMM0)); + + // OR the sign bit in. + MOVSD(XMM0, R(src)); + PAND(XMM0, M(&double_sign_bit)); + PSRLQ(XMM0, 32); + POR(XMM1, R(XMM0)); + + FixupBranch end = J(false); // Goto end + + SetJumpTarget(NoDenormalize); + SetJumpTarget(NoDenormalize2); + + // Don't Denormalize + + // We want bits 0, 1 + MOVSD(XMM1, R(src)); + PAND(XMM1, M(&double_top_two_bits)); + PSRLQ(XMM1, 32); + + // And 5 through to 34 + MOVSD(XMM0, R(src)); + PAND(XMM0, M(&double_bottom_bits)); + PSRLQ(XMM0, 29); + + // OR them togther + POR(XMM1, R(XMM0)); + + // End + SetJumpTarget(end); + MOVDDUP(dst, R(XMM1)); +} + +#else // MORE_ACCURATE_DOUBLETOSINGLE + +alignas(16) static const __m128i double_sign_bit = _mm_set_epi64x(0xffffffffffffffff, + 0x7fffffffffffffff); +alignas(16) static const __m128i single_qnan_bit = _mm_set_epi64x(0xffffffffffffffff, + 0xffffffffffbfffff); +alignas(16) static const __m128i double_qnan_bit = _mm_set_epi64x(0xffffffffffffffff, + 0xfff7ffffffffffff); + +// Smallest positive double that results in a normalized single. +alignas(16) static const double min_norm_single = std::numeric_limits::min(); + +void EmuCodeBlock::ConvertDoubleToSingle(X64Reg dst, X64Reg src) +{ + // Most games have flush-to-zero enabled, which causes the single -> double -> single process here + // to be lossy. + // This is a problem when games use float operations to copy non-float data. + // Changing the FPU mode is very expensive, so we can't do that. + // Here, check to see if the source is small enough that it will result in a denormal, and pass it + // to the x87 unit + // if it is. + avx_op(&XEmitter::VPAND, &XEmitter::PAND, XMM0, R(src), M(&double_sign_bit), true, true); + UCOMISD(XMM0, M(&min_norm_single)); + FixupBranch nanConversion = J_CC(CC_P, true); + FixupBranch denormalConversion = J_CC(CC_B, true); + CVTSD2SS(dst, R(src)); + + SwitchToFarCode(); + SetJumpTarget(nanConversion); + MOVQ_xmm(R(RSCRATCH), src); + // Put the quiet bit into CF. + BT(64, R(RSCRATCH), Imm8(51)); + CVTSD2SS(dst, R(src)); + FixupBranch continue1 = J_CC(CC_C, true); + // Clear the quiet bit of the SNaN, which was 0 (signalling) but got set to 1 (quiet) by + // conversion. + ANDPS(dst, M(&single_qnan_bit)); + FixupBranch continue2 = J(true); + + SetJumpTarget(denormalConversion); + MOVSD(M(&temp64), src); + FLD(64, M(&temp64)); + FSTP(32, M(&temp32)); + MOVSS(dst, M(&temp32)); + FixupBranch continue3 = J(true); + SwitchToNearCode(); + + SetJumpTarget(continue1); + SetJumpTarget(continue2); + SetJumpTarget(continue3); + // We'd normally need to MOVDDUP here to put the single in the top half of the output register + // too, but + // this function is only used to go directly to a following store, so we omit the MOVDDUP here. +} +#endif // MORE_ACCURATE_DOUBLETOSINGLE + +// Converting single->double is a bit easier because all single denormals are double normals. +void EmuCodeBlock::ConvertSingleToDouble(X64Reg dst, X64Reg src, bool src_is_gpr) +{ + X64Reg gprsrc = src_is_gpr ? src : RSCRATCH; + if (src_is_gpr) + { + MOVD_xmm(dst, R(src)); + } + else + { + if (dst != src) + MOVAPS(dst, R(src)); + MOVD_xmm(R(RSCRATCH), src); + } + + UCOMISS(dst, R(dst)); + CVTSS2SD(dst, R(dst)); + FixupBranch nanConversion = J_CC(CC_P, true); + + SwitchToFarCode(); + SetJumpTarget(nanConversion); + TEST(32, R(gprsrc), Imm32(0x00400000)); + FixupBranch continue1 = J_CC(CC_NZ, true); + ANDPD(dst, M(&double_qnan_bit)); + FixupBranch continue2 = J(true); + SwitchToNearCode(); + + SetJumpTarget(continue1); + SetJumpTarget(continue2); + MOVDDUP(dst, R(dst)); +} + +alignas(16) static const u64 psDoubleExp[2] = {0x7FF0000000000000ULL, 0}; +alignas(16) static const u64 psDoubleFrac[2] = {0x000FFFFFFFFFFFFFULL, 0}; +alignas(16) static const u64 psDoubleNoSign[2] = {0x7FFFFFFFFFFFFFFFULL, 0}; + +// TODO: it might be faster to handle FPRF in the same way as CR is currently handled for integer, +// storing +// the result of each floating point op and calculating it when needed. This is trickier than for +// integers +// though, because there's 32 possible FPRF bit combinations but only 9 categories of floating point +// values, +// which makes the whole thing rather trickier. +// Fortunately, PPCAnalyzer can optimize out a large portion of FPRF calculations, so maybe this +// isn't +// quite that necessary. +void EmuCodeBlock::SetFPRF(Gen::X64Reg xmm) +{ + AND(32, PPCSTATE(fpscr), Imm32(~FPRF_MASK)); + + FixupBranch continue1, continue2, continue3, continue4; + if (cpu_info.bSSE4_1) + { + MOVQ_xmm(R(RSCRATCH), xmm); + SHR(64, R(RSCRATCH), Imm8(63)); // Get the sign bit; almost all the branches need it. + PTEST(xmm, M(psDoubleExp)); + FixupBranch maxExponent = J_CC(CC_C); + FixupBranch zeroExponent = J_CC(CC_Z); + + // Nice normalized number: sign ? PPC_FPCLASS_NN : PPC_FPCLASS_PN; + LEA(32, RSCRATCH, MScaled(RSCRATCH, MathUtil::PPC_FPCLASS_NN - MathUtil::PPC_FPCLASS_PN, + MathUtil::PPC_FPCLASS_PN)); + continue1 = J(); + + SetJumpTarget(maxExponent); + PTEST(xmm, M(psDoubleFrac)); + FixupBranch notNAN = J_CC(CC_Z); + + // Max exponent + mantissa: PPC_FPCLASS_QNAN + MOV(32, R(RSCRATCH), Imm32(MathUtil::PPC_FPCLASS_QNAN)); + continue2 = J(); + + // Max exponent + no mantissa: sign ? PPC_FPCLASS_NINF : PPC_FPCLASS_PINF; + SetJumpTarget(notNAN); + LEA(32, RSCRATCH, MScaled(RSCRATCH, MathUtil::PPC_FPCLASS_NINF - MathUtil::PPC_FPCLASS_PINF, + MathUtil::PPC_FPCLASS_PINF)); + continue3 = J(); + + SetJumpTarget(zeroExponent); + PTEST(xmm, R(xmm)); + FixupBranch zero = J_CC(CC_Z); + + // No exponent + mantissa: sign ? PPC_FPCLASS_ND : PPC_FPCLASS_PD; + LEA(32, RSCRATCH, MScaled(RSCRATCH, MathUtil::PPC_FPCLASS_ND - MathUtil::PPC_FPCLASS_PD, + MathUtil::PPC_FPCLASS_PD)); + continue4 = J(); + + // Zero: sign ? PPC_FPCLASS_NZ : PPC_FPCLASS_PZ; + SetJumpTarget(zero); + SHL(32, R(RSCRATCH), Imm8(4)); + ADD(32, R(RSCRATCH), Imm8(MathUtil::PPC_FPCLASS_PZ)); + } + else + { + MOVQ_xmm(R(RSCRATCH), xmm); + TEST(64, R(RSCRATCH), M(psDoubleExp)); + FixupBranch zeroExponent = J_CC(CC_Z); + AND(64, R(RSCRATCH), M(psDoubleNoSign)); + CMP(64, R(RSCRATCH), M(psDoubleExp)); + FixupBranch nan = + J_CC(CC_G); // This works because if the sign bit is set, RSCRATCH is negative + FixupBranch infinity = J_CC(CC_E); + MOVQ_xmm(R(RSCRATCH), xmm); + SHR(64, R(RSCRATCH), Imm8(63)); + LEA(32, RSCRATCH, MScaled(RSCRATCH, MathUtil::PPC_FPCLASS_NN - MathUtil::PPC_FPCLASS_PN, + MathUtil::PPC_FPCLASS_PN)); + continue1 = J(); + SetJumpTarget(nan); + MOVQ_xmm(R(RSCRATCH), xmm); + SHR(64, R(RSCRATCH), Imm8(63)); + MOV(32, R(RSCRATCH), Imm32(MathUtil::PPC_FPCLASS_QNAN)); + continue2 = J(); + SetJumpTarget(infinity); + MOVQ_xmm(R(RSCRATCH), xmm); + SHR(64, R(RSCRATCH), Imm8(63)); + LEA(32, RSCRATCH, MScaled(RSCRATCH, MathUtil::PPC_FPCLASS_NINF - MathUtil::PPC_FPCLASS_PINF, + MathUtil::PPC_FPCLASS_PINF)); + continue3 = J(); + SetJumpTarget(zeroExponent); + TEST(64, R(RSCRATCH), R(RSCRATCH)); + FixupBranch zero = J_CC(CC_Z); + SHR(64, R(RSCRATCH), Imm8(63)); + LEA(32, RSCRATCH, MScaled(RSCRATCH, MathUtil::PPC_FPCLASS_ND - MathUtil::PPC_FPCLASS_PD, + MathUtil::PPC_FPCLASS_PD)); + continue4 = J(); + SetJumpTarget(zero); + SHR(64, R(RSCRATCH), Imm8(63)); + SHL(32, R(RSCRATCH), Imm8(4)); + ADD(32, R(RSCRATCH), Imm8(MathUtil::PPC_FPCLASS_PZ)); + } + + SetJumpTarget(continue1); + SetJumpTarget(continue2); + SetJumpTarget(continue3); + SetJumpTarget(continue4); + SHL(32, R(RSCRATCH), Imm8(FPRF_SHIFT)); + OR(32, PPCSTATE(fpscr), R(RSCRATCH)); +} + +void EmuCodeBlock::JitGetAndClearCAOV(bool oe) +{ + if (oe) + AND(8, PPCSTATE(xer_so_ov), Imm8(~XER_OV_MASK)); // XER.OV = 0 + SHR(8, PPCSTATE(xer_ca), Imm8(1)); // carry = XER.CA, XER.CA = 0 +} + +void EmuCodeBlock::JitSetCA() +{ + MOV(8, PPCSTATE(xer_ca), Imm8(1)); // XER.CA = 1 +} + +// Some testing shows CA is set roughly ~1/3 of the time (relative to clears), so +// branchless calculation of CA is probably faster in general. +void EmuCodeBlock::JitSetCAIf(CCFlags conditionCode) +{ + SETcc(conditionCode, PPCSTATE(xer_ca)); +} + +void EmuCodeBlock::JitClearCA() +{ + MOV(8, PPCSTATE(xer_ca), Imm8(0)); +} + +void EmuCodeBlock::Clear() +{ + backPatchInfo.clear(); + exceptionHandlerAtLoc.clear(); +} diff --git a/Source/Core/Core/PowerPC/Jit64Common/Jit64Util.h b/Source/Core/Core/PowerPC/Jit64Common/Jit64Util.h new file mode 100644 index 0000000000..6c506d1bf2 --- /dev/null +++ b/Source/Core/Core/PowerPC/Jit64Common/Jit64Util.h @@ -0,0 +1,212 @@ +// Copyright 2010 Dolphin Emulator Project +// Licensed under GPLv2+ +// Refer to the license.txt file included./ + +#pragma once + +#include + +#include "Common/BitSet.h" +#include "Common/CPUDetect.h" +#include "Common/CommonTypes.h" +#include "Common/x64Emitter.h" +#include "Core/PowerPC/PowerPC.h" + +namespace MMIO +{ +class Mapping; +} + +// We offset by 0x80 because the range of one byte memory offsets is +// -0x80..0x7f. +#define PPCSTATE(x) \ + MDisp(RPPCSTATE, (int)((char*)&PowerPC::ppcState.x - (char*)&PowerPC::ppcState) - 0x80) +// In case you want to disable the ppcstate register: +// #define PPCSTATE(x) M(&PowerPC::ppcState.x) +#define PPCSTATE_LR PPCSTATE(spr[SPR_LR]) +#define PPCSTATE_CTR PPCSTATE(spr[SPR_CTR]) +#define PPCSTATE_SRR0 PPCSTATE(spr[SPR_SRR0]) +#define PPCSTATE_SRR1 PPCSTATE(spr[SPR_SRR1]) + +// A place to throw blocks of code we don't want polluting the cache, e.g. rarely taken +// exception branches. +class FarCodeCache : public Gen::X64CodeBlock +{ +private: + bool m_enabled = false; + +public: + bool Enabled() const { return m_enabled; } + void Init(int size) + { + AllocCodeSpace(size); + m_enabled = true; + } + void Shutdown() + { + FreeCodeSpace(); + m_enabled = false; + } +}; + +static const int CODE_SIZE = 1024 * 1024 * 32; + +// a bit of a hack; the MMU results in a vast amount more code ending up in the far cache, +// mostly exception handling, so give it a whole bunch more space if the MMU is on. +static const int FARCODE_SIZE = 1024 * 1024 * 8; +static const int FARCODE_SIZE_MMU = 1024 * 1024 * 48; + +// same for the trampoline code cache, because fastmem results in far more backpatches in MMU mode +static const int TRAMPOLINE_CODE_SIZE = 1024 * 1024 * 8; +static const int TRAMPOLINE_CODE_SIZE_MMU = 1024 * 1024 * 32; + +// Stores information we need to batch-patch a MOV with a call to the slow read/write path after +// it faults. There will be 10s of thousands of these structs live, so be wary of making this too +// big. +struct TrampolineInfo final +{ + // The start of the store operation that failed -- we will patch a JMP here + u8* start; + + // The start + len = end of the store operation (points to the next instruction) + u32 len; + + // The PPC PC for the current load/store block + u32 pc; + + // Saved because we need these to make the ABI call in the trampoline + BitSet32 registersInUse; + + // The MOV operation + Gen::X64Reg nonAtomicSwapStoreSrc; + + // src/dest for load/store + s32 offset; + Gen::X64Reg op_reg; + Gen::OpArg op_arg; + + // Original SafeLoadXXX/SafeStoreXXX flags + u8 flags; + + // Memory access size (in bytes) + u8 accessSize : 4; + + // true if this is a read op vs a write + bool read : 1; + + // for read operations, true if needs sign-extension after load + bool signExtend : 1; + + // Set to true if we added the offset to the address and need to undo it + bool offsetAddedToAddress : 1; +}; + +// Like XCodeBlock but has some utilities for memory access. +class EmuCodeBlock : public Gen::X64CodeBlock +{ +public: + FarCodeCache farcode; + u8* nearcode; // Backed up when we switch to far code. + + void MemoryExceptionCheck(); + + // Simple functions to switch between near and far code emitting + void SwitchToFarCode() + { + nearcode = GetWritableCodePtr(); + SetCodePtr(farcode.GetWritableCodePtr()); + } + + void SwitchToNearCode() + { + farcode.SetCodePtr(GetWritableCodePtr()); + SetCodePtr(nearcode); + } + + Gen::FixupBranch CheckIfSafeAddress(const Gen::OpArg& reg_value, Gen::X64Reg reg_addr, + BitSet32 registers_in_use); + void UnsafeLoadRegToReg(Gen::X64Reg reg_addr, Gen::X64Reg reg_value, int accessSize, + s32 offset = 0, bool signExtend = false); + void UnsafeLoadRegToRegNoSwap(Gen::X64Reg reg_addr, Gen::X64Reg reg_value, int accessSize, + s32 offset, bool signExtend = false); + // these return the address of the MOV, for backpatching + void UnsafeWriteRegToReg(Gen::OpArg reg_value, Gen::X64Reg reg_addr, int accessSize, + s32 offset = 0, bool swap = true, Gen::MovInfo* info = nullptr); + void UnsafeWriteRegToReg(Gen::X64Reg reg_value, Gen::X64Reg reg_addr, int accessSize, + s32 offset = 0, bool swap = true, Gen::MovInfo* info = nullptr) + { + UnsafeWriteRegToReg(R(reg_value), reg_addr, accessSize, offset, swap, info); + } + bool UnsafeLoadToReg(Gen::X64Reg reg_value, Gen::OpArg opAddress, int accessSize, s32 offset, + bool signExtend, Gen::MovInfo* info = nullptr); + void UnsafeWriteGatherPipe(int accessSize); + + // Generate a load/write from the MMIO handler for a given address. Only + // call for known addresses in MMIO range (MMIO::IsMMIOAddress). + void MMIOLoadToReg(MMIO::Mapping* mmio, Gen::X64Reg reg_value, BitSet32 registers_in_use, + u32 address, int access_size, bool sign_extend); + + enum SafeLoadStoreFlags + { + SAFE_LOADSTORE_NO_SWAP = 1, + SAFE_LOADSTORE_NO_PROLOG = 2, + // This indicates that the write being generated cannot be patched (and thus can't use fastmem) + SAFE_LOADSTORE_NO_FASTMEM = 4, + SAFE_LOADSTORE_CLOBBER_RSCRATCH_INSTEAD_OF_ADDR = 8, + // Force slowmem (used when generating fallbacks in trampolines) + SAFE_LOADSTORE_FORCE_SLOWMEM = 16, + SAFE_LOADSTORE_DR_ON = 32, + }; + + void SafeLoadToReg(Gen::X64Reg reg_value, const Gen::OpArg& opAddress, int accessSize, s32 offset, + BitSet32 registersInUse, bool signExtend, int flags = 0); + void SafeLoadToRegImmediate(Gen::X64Reg reg_value, u32 address, int accessSize, + BitSet32 registersInUse, bool signExtend); + + // Clobbers RSCRATCH or reg_addr depending on the relevant flag. Preserves + // reg_value if the load fails and js.memcheck is enabled. + // Works with immediate inputs and simple registers only. + void SafeWriteRegToReg(Gen::OpArg reg_value, Gen::X64Reg reg_addr, int accessSize, s32 offset, + BitSet32 registersInUse, int flags = 0); + void SafeWriteRegToReg(Gen::X64Reg reg_value, Gen::X64Reg reg_addr, int accessSize, s32 offset, + BitSet32 registersInUse, int flags = 0) + { + SafeWriteRegToReg(R(reg_value), reg_addr, accessSize, offset, registersInUse, flags); + } + + // applies to safe and unsafe WriteRegToReg + bool WriteClobbersRegValue(int accessSize, bool swap) + { + return swap && !cpu_info.bMOVBE && accessSize > 8; + } + + void WriteToConstRamAddress(int accessSize, Gen::OpArg arg, u32 address, bool swap = true); + // returns true if an exception could have been caused + bool WriteToConstAddress(int accessSize, Gen::OpArg arg, u32 address, BitSet32 registersInUse); + void JitGetAndClearCAOV(bool oe); + void JitSetCA(); + void JitSetCAIf(Gen::CCFlags conditionCode); + void JitClearCA(); + + void avx_op(void (Gen::XEmitter::*avxOp)(Gen::X64Reg, Gen::X64Reg, const Gen::OpArg&), + void (Gen::XEmitter::*sseOp)(Gen::X64Reg, const Gen::OpArg&), Gen::X64Reg regOp, + const Gen::OpArg& arg1, const Gen::OpArg& arg2, bool packed = true, + bool reversible = false); + void avx_op(void (Gen::XEmitter::*avxOp)(Gen::X64Reg, Gen::X64Reg, const Gen::OpArg&, u8), + void (Gen::XEmitter::*sseOp)(Gen::X64Reg, const Gen::OpArg&, u8), Gen::X64Reg regOp, + const Gen::OpArg& arg1, const Gen::OpArg& arg2, u8 imm); + + void ForceSinglePrecision(Gen::X64Reg output, const Gen::OpArg& input, bool packed = true, + bool duplicate = false); + void Force25BitPrecision(Gen::X64Reg output, const Gen::OpArg& input, Gen::X64Reg tmp); + + // RSCRATCH might get trashed + void ConvertSingleToDouble(Gen::X64Reg dst, Gen::X64Reg src, bool src_is_gpr = false); + void ConvertDoubleToSingle(Gen::X64Reg dst, Gen::X64Reg src); + void SetFPRF(Gen::X64Reg xmm); + void Clear(); + +protected: + std::unordered_map backPatchInfo; + std::unordered_map exceptionHandlerAtLoc; +}; diff --git a/Source/Core/Core/PowerPC/Jit64Common/TrampolineCache.cpp b/Source/Core/Core/PowerPC/Jit64Common/TrampolineCache.cpp new file mode 100644 index 0000000000..67a2e4a9e5 --- /dev/null +++ b/Source/Core/Core/PowerPC/Jit64Common/TrampolineCache.cpp @@ -0,0 +1,86 @@ +// Copyright 2014 Dolphin Emulator Project +// Licensed under GPLv2+ +// Refer to the license.txt file included. + +#include "Core/PowerPC/Jit64Common/TrampolineCache.h" + +#include +#include + +#include "Common/CommonTypes.h" +#include "Common/JitRegister.h" +#include "Common/MsgHandler.h" +#include "Common/x64Emitter.h" +#include "Core/PowerPC/Jit64Common/Jit64Base.h" +#include "Core/PowerPC/Jit64Common/Jit64Util.h" +#include "Core/PowerPC/JitCommon/JitBase.h" +#include "Core/PowerPC/PowerPC.h" + +#ifdef _WIN32 +#include +#endif + +using namespace Gen; + +void TrampolineCache::Init(int size) +{ + AllocCodeSpace(size); +} + +void TrampolineCache::ClearCodeSpace() +{ + X64CodeBlock::ClearCodeSpace(); +} + +void TrampolineCache::Shutdown() +{ + FreeCodeSpace(); +} + +const u8* TrampolineCache::GenerateTrampoline(const TrampolineInfo& info) +{ + if (info.read) + { + return GenerateReadTrampoline(info); + } + + return GenerateWriteTrampoline(info); +} + +const u8* TrampolineCache::GenerateReadTrampoline(const TrampolineInfo& info) +{ + if (GetSpaceLeft() < 1024) + PanicAlert("Trampoline cache full"); + + const u8* trampoline = GetCodePtr(); + + SafeLoadToReg(info.op_reg, info.op_arg, info.accessSize << 3, info.offset, info.registersInUse, + info.signExtend, info.flags | SAFE_LOADSTORE_FORCE_SLOWMEM); + + JMP(info.start + info.len, true); + + JitRegister::Register(trampoline, GetCodePtr(), "JIT_ReadTrampoline_%x", info.pc); + return trampoline; +} + +const u8* TrampolineCache::GenerateWriteTrampoline(const TrampolineInfo& info) +{ + if (GetSpaceLeft() < 1024) + PanicAlert("Trampoline cache full"); + + const u8* trampoline = GetCodePtr(); + + // Don't treat FIFO writes specially for now because they require a burst + // check anyway. + + // PC is used by memory watchpoints (if enabled) or to print accurate PC locations in debug logs + MOV(32, PPCSTATE(pc), Imm32(info.pc)); + + SafeWriteRegToReg(info.op_arg, info.op_reg, info.accessSize << 3, info.offset, + info.registersInUse, info.flags | SAFE_LOADSTORE_FORCE_SLOWMEM); + + JMP(info.start + info.len, true); + + JitRegister::Register(trampoline, GetCodePtr(), "JIT_WriteTrampoline_%x", info.pc); + return trampoline; +} diff --git a/Source/Core/Core/PowerPC/Jit64Common/TrampolineCache.h b/Source/Core/Core/PowerPC/Jit64Common/TrampolineCache.h new file mode 100644 index 0000000000..7e4cac8b2f --- /dev/null +++ b/Source/Core/Core/PowerPC/Jit64Common/TrampolineCache.h @@ -0,0 +1,23 @@ +// Copyright 2014 Dolphin Emulator Project +// Licensed under GPLv2+ +// Refer to the license.txt file included. + +#pragma once + +#include "Common/CommonTypes.h" +#include "Core/PowerPC/Jit64Common/Jit64Util.h" + +// We need at least this many bytes for backpatching. +const int BACKPATCH_SIZE = 5; + +class TrampolineCache : public EmuCodeBlock +{ + const u8* GenerateReadTrampoline(const TrampolineInfo& info); + const u8* GenerateWriteTrampoline(const TrampolineInfo& info); + +public: + void Init(int size); + void Shutdown(); + const u8* GenerateTrampoline(const TrampolineInfo& info); + void ClearCodeSpace(); +}; diff --git a/Source/Core/Core/PowerPC/Jit64IL/JitIL.h b/Source/Core/Core/PowerPC/Jit64IL/JitIL.h index 90f0f95a49..5dd943aa02 100644 --- a/Source/Core/Core/PowerPC/Jit64IL/JitIL.h +++ b/Source/Core/Core/PowerPC/Jit64IL/JitIL.h @@ -21,7 +21,6 @@ #include "Common/x64Emitter.h" #include "Core/PowerPC/Gekko.h" #include "Core/PowerPC/Jit64/JitAsm.h" -#include "Core/PowerPC/JitCommon/JitBase.h" #include "Core/PowerPC/JitCommon/JitCache.h" #include "Core/PowerPC/JitILCommon/JitILBase.h" #include "Core/PowerPC/PPCAnalyst.h" diff --git a/Source/Core/Core/PowerPC/JitArm64/Jit.h b/Source/Core/Core/PowerPC/JitArm64/Jit.h index 3393d12b0d..bb8482aece 100644 --- a/Source/Core/Core/PowerPC/JitArm64/Jit.h +++ b/Source/Core/Core/PowerPC/JitArm64/Jit.h @@ -17,6 +17,9 @@ #include "Core/PowerPC/JitCommon/JitBase.h" #include "Core/PowerPC/PPCAnalyst.h" +constexpr int CODE_SIZE = 1024 * 1024 * 32; +constexpr int FARCODE_SIZE_MMU = 1024 * 1024 * 48; + class JitArm64 : public JitBase, public Arm64Gen::ARM64CodeBlock, public CommonAsmRoutinesBase { public: diff --git a/Source/Core/Core/PowerPC/JitCommon/JitBackpatch.cpp b/Source/Core/Core/PowerPC/JitCommon/JitBackpatch.cpp deleted file mode 100644 index 83119d2189..0000000000 --- a/Source/Core/Core/PowerPC/JitCommon/JitBackpatch.cpp +++ /dev/null @@ -1,124 +0,0 @@ -// Copyright 2008 Dolphin Emulator Project -// Licensed under GPLv2+ -// Refer to the license.txt file included. - -#include -#include - -#include "disasm.h" - -#include "Common/Assert.h" -#include "Common/BitSet.h" -#include "Common/CommonFuncs.h" -#include "Common/CommonTypes.h" -#include "Common/MsgHandler.h" -#include "Common/x64Emitter.h" -#include "Core/HW/Memmap.h" -#include "Core/PowerPC/JitCommon/JitBase.h" - -using namespace Gen; - -// This generates some fairly heavy trampolines, but it doesn't really hurt. -// Only instructions that access I/O will get these, and there won't be that -// many of them in a typical program/game. -bool Jitx86Base::HandleFault(uintptr_t access_address, SContext* ctx) -{ - // TODO: do we properly handle off-the-end? - if (access_address >= (uintptr_t)Memory::physical_base && - access_address < (uintptr_t)Memory::physical_base + 0x100010000) - return BackPatch((u32)(access_address - (uintptr_t)Memory::physical_base), ctx); - if (access_address >= (uintptr_t)Memory::logical_base && - access_address < (uintptr_t)Memory::logical_base + 0x100010000) - return BackPatch((u32)(access_address - (uintptr_t)Memory::logical_base), ctx); - - return false; -} - -bool Jitx86Base::BackPatch(u32 emAddress, SContext* ctx) -{ - u8* codePtr = (u8*)ctx->CTX_PC; - - if (!IsInSpace(codePtr)) - return false; // this will become a regular crash real soon after this - - auto it = backPatchInfo.find(codePtr); - if (it == backPatchInfo.end()) - { - PanicAlert("BackPatch: no register use entry for address %p", codePtr); - return false; - } - - TrampolineInfo& info = it->second; - - u8* exceptionHandler = nullptr; - if (jit->jo.memcheck) - { - auto it2 = exceptionHandlerAtLoc.find(codePtr); - if (it2 != exceptionHandlerAtLoc.end()) - exceptionHandler = it2->second; - } - - // In the trampoline code, we jump back into the block at the beginning - // of the next instruction. The next instruction comes immediately - // after the backpatched operation, or BACKPATCH_SIZE bytes after the start - // of the backpatched operation, whichever comes last. (The JIT inserts NOPs - // into the original code if necessary to ensure there is enough space - // to insert the backpatch jump.) - - jit->js.generatingTrampoline = true; - jit->js.trampolineExceptionHandler = exceptionHandler; - - // Generate the trampoline. - const u8* trampoline = trampolines.GenerateTrampoline(info); - jit->js.generatingTrampoline = false; - jit->js.trampolineExceptionHandler = nullptr; - - u8* start = info.start; - - // Patch the original memory operation. - XEmitter emitter(start); - emitter.JMP(trampoline, true); - // NOPs become dead code - const u8* end = info.start + info.len; - for (const u8* i = emitter.GetCodePtr(); i < end; ++i) - emitter.INT3(); - - // Rewind time to just before the start of the write block. If we swapped memory - // before faulting (eg: the store+swap was not an atomic op like MOVBE), let's - // swap it back so that the swap can happen again (this double swap isn't ideal but - // only happens the first time we fault). - if (info.nonAtomicSwapStoreSrc != INVALID_REG) - { - u64* ptr = ContextRN(ctx, info.nonAtomicSwapStoreSrc); - switch (info.accessSize << 3) - { - case 8: - // No need to swap a byte - break; - case 16: - *ptr = Common::swap16(static_cast(*ptr)); - break; - case 32: - *ptr = Common::swap32(static_cast(*ptr)); - break; - case 64: - *ptr = Common::swap64(static_cast(*ptr)); - break; - default: - _dbg_assert_(DYNA_REC, 0); - break; - } - } - - // This is special code to undo the LEA in SafeLoadToReg if it clobbered the address - // register in the case where reg_value shared the same location as opAddress. - if (info.offsetAddedToAddress) - { - u64* ptr = ContextRN(ctx, info.op_arg.GetSimpleReg()); - *ptr -= static_cast(info.offset); - } - - ctx->CTX_PC = reinterpret_cast(trampoline); - - return true; -} diff --git a/Source/Core/Core/PowerPC/JitCommon/JitBase.cpp b/Source/Core/Core/PowerPC/JitCommon/JitBase.cpp index 231087bf21..2b0e98aaed 100644 --- a/Source/Core/Core/PowerPC/JitCommon/JitBase.cpp +++ b/Source/Core/Core/PowerPC/JitCommon/JitBase.cpp @@ -2,18 +2,11 @@ // Licensed under GPLv2+ // Refer to the license.txt file included. -#include -#include - -#include "disasm.h" +#include "Core/PowerPC/JitCommon/JitBase.h" #include "Common/CommonTypes.h" -#include "Common/GekkoDisassembler.h" -#include "Common/Logging/Log.h" -#include "Common/StringUtil.h" #include "Core/ConfigManager.h" #include "Core/HW/CPU.h" -#include "Core/PowerPC/JitCommon/JitBase.h" #include "Core/PowerPC/PPCAnalyst.h" #include "Core/PowerPC/PowerPC.h" @@ -30,44 +23,6 @@ u32 Helper_Mask(u8 mb, u8 me) return mb > me ? ~mask : mask; } -void LogGeneratedX86(int size, PPCAnalyst::CodeBuffer* code_buffer, const u8* normalEntry, - JitBlock* b) -{ - for (int i = 0; i < size; i++) - { - const PPCAnalyst::CodeOp& op = code_buffer->codebuffer[i]; - std::string temp = StringFromFormat( - "%08x %s", op.address, GekkoDisassembler::Disassemble(op.inst.hex, op.address).c_str()); - DEBUG_LOG(DYNA_REC, "IR_X86 PPC: %s\n", temp.c_str()); - } - - disassembler x64disasm; - x64disasm.set_syntax_intel(); - - u64 disasmPtr = (u64)normalEntry; - const u8* end = normalEntry + b->codeSize; - - while ((u8*)disasmPtr < end) - { - char sptr[1000] = ""; - disasmPtr += x64disasm.disasm64(disasmPtr, disasmPtr, (u8*)disasmPtr, sptr); - DEBUG_LOG(DYNA_REC, "IR_X86 x86: %s", sptr); - } - - if (b->codeSize <= 250) - { - std::stringstream ss; - ss << std::hex; - for (u8 i = 0; i <= b->codeSize; i++) - { - ss.width(2); - ss.fill('0'); - ss << (u32) * (normalEntry + i); - } - DEBUG_LOG(DYNA_REC, "IR_X86 bin: %s\n\n\n", ss.str().c_str()); - } -} - bool JitBase::MergeAllowedNextInstructions(int count) { if (CPU::GetState() == CPU::CPU_STEPPING || js.instructionsLeft < count) diff --git a/Source/Core/Core/PowerPC/JitCommon/JitBase.h b/Source/Core/Core/PowerPC/JitCommon/JitBase.h index 611395e3b8..bd7bf2c2ae 100644 --- a/Source/Core/Core/PowerPC/JitCommon/JitBase.h +++ b/Source/Core/Core/PowerPC/JitCommon/JitBase.h @@ -16,27 +16,10 @@ #include "Core/ConfigManager.h" #include "Core/MachineContext.h" #include "Core/PowerPC/CPUCoreBase.h" -#include "Core/PowerPC/Jit64Common/Jit64AsmCommon.h" +#include "Core/PowerPC/JitCommon/JitAsmCommon.h" #include "Core/PowerPC/JitCommon/JitCache.h" -#include "Core/PowerPC/JitCommon/Jit_Util.h" -#include "Core/PowerPC/JitCommon/TrampolineCache.h" #include "Core/PowerPC/PPCAnalyst.h" -// TODO: find a better place for x86-specific stuff -// The following register assignments are common to Jit64 and Jit64IL: -// RSCRATCH and RSCRATCH2 are always scratch registers and can be used without -// limitation. -#define RSCRATCH RAX -#define RSCRATCH2 RDX -// RSCRATCH_EXTRA may be in the allocation order, so it has to be flushed -// before use. -#define RSCRATCH_EXTRA RCX -// RMEM points to the start of emulated memory. -#define RMEM RBX -// RPPCSTATE points to ppcState + 0x80. It's offset because we want to be able -// to address as much as possible in a one-byte offset form. -#define RPPCSTATE RBP - // Use these to control the instruction selection // #define INSTRUCTION_START FallBackToInterpreter(inst); return; // #define INSTRUCTION_START PPCTables::CountInstruction(inst); @@ -142,21 +125,7 @@ public: virtual bool HandleStackFault() { return false; } }; -class Jitx86Base : public JitBase, public QuantizedMemoryRoutines -{ -protected: - bool BackPatch(u32 emAddress, SContext* ctx); - JitBlockCache blocks; - TrampolineCache trampolines; - -public: - JitBlockCache* GetBlockCache() override { return &blocks; } - bool HandleFault(uintptr_t access_address, SContext* ctx) override; -}; - void Jit(u32 em_address); // Merged routines that should be moved somewhere better u32 Helper_Mask(u8 mb, u8 me); -void LogGeneratedX86(int size, PPCAnalyst::CodeBuffer* code_buffer, const u8* normalEntry, - JitBlock* b); diff --git a/Source/Core/Core/PowerPC/JitCommon/JitCache.cpp b/Source/Core/Core/PowerPC/JitCommon/JitCache.cpp index 1d601b3a64..29fd7f2769 100644 --- a/Source/Core/Core/PowerPC/JitCommon/JitCache.cpp +++ b/Source/Core/Core/PowerPC/JitCommon/JitCache.cpp @@ -352,34 +352,3 @@ void JitBaseBlockCache::InvalidateICache(u32 address, const u32 length, bool for } } } - -void JitBlockCache::WriteLinkBlock(const JitBlock::LinkData& source, const JitBlock* dest) -{ - u8* location = source.exitPtrs; - const u8* address = dest ? dest->checkedEntry : jit->GetAsmRoutines()->dispatcher; - XEmitter emit(location); - if (*location == 0xE8) - { - emit.CALL(address); - } - else - { - // If we're going to link with the next block, there is no need - // to emit JMP. So just NOP out the gap to the next block. - // Support up to 3 additional bytes because of alignment. - s64 offset = address - emit.GetCodePtr(); - if (offset > 0 && offset <= 5 + 3) - emit.NOP(offset); - else - emit.JMP(address, true); - } -} - -void JitBlockCache::WriteDestroyBlock(const JitBlock& block) -{ - // Only clear the entry points as we might still be within this block. - XEmitter emit((u8*)block.checkedEntry); - emit.INT3(); - XEmitter emit2((u8*)block.normalEntry); - emit2.INT3(); -} diff --git a/Source/Core/Core/PowerPC/JitCommon/JitCache.h b/Source/Core/Core/PowerPC/JitCommon/JitCache.h index f3a3c8b4a3..0694400177 100644 --- a/Source/Core/Core/PowerPC/JitCommon/JitCache.h +++ b/Source/Core/Core/PowerPC/JitCommon/JitCache.h @@ -186,11 +186,3 @@ public: u32* GetBlockBitSet() const { return valid_block.m_valid_block.get(); } }; - -// x86 BlockCache -class JitBlockCache : public JitBaseBlockCache -{ -private: - void WriteLinkBlock(const JitBlock::LinkData& source, const JitBlock* dest) override; - void WriteDestroyBlock(const JitBlock& block) override; -}; diff --git a/Source/Core/Core/PowerPC/JitCommon/Jit_Util.cpp b/Source/Core/Core/PowerPC/JitCommon/Jit_Util.cpp deleted file mode 100644 index 1f44650ef4..0000000000 --- a/Source/Core/Core/PowerPC/JitCommon/Jit_Util.cpp +++ /dev/null @@ -1,1113 +0,0 @@ -// Copyright 2008 Dolphin Emulator Project -// Licensed under GPLv2+ -// Refer to the license.txt file included. - -#include "Core/PowerPC/JitCommon/Jit_Util.h" -#include "Common/BitSet.h" -#include "Common/CommonTypes.h" -#include "Common/Intrinsics.h" -#include "Common/MathUtil.h" -#include "Common/x64ABI.h" -#include "Common/x64Emitter.h" -#include "Core/HW/MMIO.h" -#include "Core/HW/Memmap.h" -#include "Core/PowerPC/JitCommon/JitBase.h" -#include "Core/PowerPC/PowerPC.h" - -using namespace Gen; - -void EmuCodeBlock::MemoryExceptionCheck() -{ - // TODO: We really should untangle the trampolines, exception handlers and - // memory checks. - - // If we are currently generating a trampoline for a failed fastmem - // load/store, the trampoline generator will have stashed the exception - // handler (that we previously generated after the fastmem instruction) in - // trampolineExceptionHandler. - if (jit->js.generatingTrampoline) - { - if (jit->js.trampolineExceptionHandler) - { - TEST(32, PPCSTATE(Exceptions), Gen::Imm32(EXCEPTION_DSI)); - J_CC(CC_NZ, jit->js.trampolineExceptionHandler); - } - return; - } - - // If memcheck (ie: MMU) mode is enabled and we haven't generated an - // exception handler for this instruction yet, we will generate an - // exception check. - if (jit->jo.memcheck && !jit->js.fastmemLoadStore && !jit->js.fixupExceptionHandler) - { - TEST(32, PPCSTATE(Exceptions), Gen::Imm32(EXCEPTION_DSI)); - jit->js.exceptionHandler = J_CC(Gen::CC_NZ, true); - jit->js.fixupExceptionHandler = true; - } -} - -void EmuCodeBlock::UnsafeLoadRegToReg(X64Reg reg_addr, X64Reg reg_value, int accessSize, s32 offset, - bool signExtend) -{ - OpArg src = MComplex(RMEM, reg_addr, SCALE_1, offset); - LoadAndSwap(accessSize, reg_value, src, signExtend); -} - -void EmuCodeBlock::UnsafeLoadRegToRegNoSwap(X64Reg reg_addr, X64Reg reg_value, int accessSize, - s32 offset, bool signExtend) -{ - if (signExtend) - MOVSX(32, accessSize, reg_value, MComplex(RMEM, reg_addr, SCALE_1, offset)); - else - MOVZX(32, accessSize, reg_value, MComplex(RMEM, reg_addr, SCALE_1, offset)); -} - -bool EmuCodeBlock::UnsafeLoadToReg(X64Reg reg_value, OpArg opAddress, int accessSize, s32 offset, - bool signExtend, MovInfo* info) -{ - bool offsetAddedToAddress = false; - OpArg memOperand; - if (opAddress.IsSimpleReg()) - { - // Deal with potential wraparound. (This is just a heuristic, and it would - // be more correct to actually mirror the first page at the end, but the - // only case where it probably actually matters is JitIL turning adds into - // offsets with the wrong sign, so whatever. Since the original code - // *could* try to wrap an address around, however, this is the correct - // place to address the issue.) - if ((u32)offset >= 0x1000) - { - // This method can potentially clobber the address if it shares a register - // with the load target. In this case we can just subtract offset from the - // register (see JitBackpatch for this implementation). - offsetAddedToAddress = (reg_value == opAddress.GetSimpleReg()); - - LEA(32, reg_value, MDisp(opAddress.GetSimpleReg(), offset)); - opAddress = R(reg_value); - offset = 0; - } - memOperand = MComplex(RMEM, opAddress.GetSimpleReg(), SCALE_1, offset); - } - else if (opAddress.IsImm()) - { - MOV(32, R(reg_value), Imm32((u32)(opAddress.Imm32() + offset))); - memOperand = MRegSum(RMEM, reg_value); - } - else - { - MOV(32, R(reg_value), opAddress); - memOperand = MComplex(RMEM, reg_value, SCALE_1, offset); - } - - LoadAndSwap(accessSize, reg_value, memOperand, signExtend, info); - return offsetAddedToAddress; -} - -// Visitor that generates code to read a MMIO value. -template -class MMIOReadCodeGenerator : public MMIO::ReadHandlingMethodVisitor -{ -public: - MMIOReadCodeGenerator(Gen::X64CodeBlock* code, BitSet32 registers_in_use, Gen::X64Reg dst_reg, - u32 address, bool sign_extend) - : m_code(code), m_registers_in_use(registers_in_use), m_dst_reg(dst_reg), m_address(address), - m_sign_extend(sign_extend) - { - } - - void VisitConstant(T value) override { LoadConstantToReg(8 * sizeof(T), value); } - void VisitDirect(const T* addr, u32 mask) override - { - LoadAddrMaskToReg(8 * sizeof(T), addr, mask); - } - void VisitComplex(const std::function* lambda) override - { - CallLambda(8 * sizeof(T), lambda); - } - -private: - // Generates code to load a constant to the destination register. In - // practice it would be better to avoid using a register for this, but it - // would require refactoring a lot of JIT code. - void LoadConstantToReg(int sbits, u32 value) - { - if (m_sign_extend) - { - u32 sign = !!(value & (1 << (sbits - 1))); - value |= sign * ((0xFFFFFFFF >> sbits) << sbits); - } - m_code->MOV(32, R(m_dst_reg), Gen::Imm32(value)); - } - - // Generate the proper MOV instruction depending on whether the read should - // be sign extended or zero extended. - void MoveOpArgToReg(int sbits, const Gen::OpArg& arg) - { - if (m_sign_extend) - m_code->MOVSX(32, sbits, m_dst_reg, arg); - else - m_code->MOVZX(32, sbits, m_dst_reg, arg); - } - - void LoadAddrMaskToReg(int sbits, const void* ptr, u32 mask) - { - m_code->MOV(64, R(RSCRATCH), ImmPtr(ptr)); - // If we do not need to mask, we can do the sign extend while loading - // from memory. If masking is required, we have to first zero extend, - // then mask, then sign extend if needed (1 instr vs. 2/3). - u32 all_ones = (1ULL << sbits) - 1; - if ((all_ones & mask) == all_ones) - { - MoveOpArgToReg(sbits, MatR(RSCRATCH)); - } - else - { - m_code->MOVZX(32, sbits, m_dst_reg, MatR(RSCRATCH)); - m_code->AND(32, R(m_dst_reg), Imm32(mask)); - if (m_sign_extend) - m_code->MOVSX(32, sbits, m_dst_reg, R(m_dst_reg)); - } - } - - void CallLambda(int sbits, const std::function* lambda) - { - m_code->ABI_PushRegistersAndAdjustStack(m_registers_in_use, 0); - m_code->ABI_CallLambdaC(lambda, m_address); - m_code->ABI_PopRegistersAndAdjustStack(m_registers_in_use, 0); - MoveOpArgToReg(sbits, R(ABI_RETURN)); - } - - Gen::X64CodeBlock* m_code; - BitSet32 m_registers_in_use; - Gen::X64Reg m_dst_reg; - u32 m_address; - bool m_sign_extend; -}; - -void EmuCodeBlock::MMIOLoadToReg(MMIO::Mapping* mmio, Gen::X64Reg reg_value, - BitSet32 registers_in_use, u32 address, int access_size, - bool sign_extend) -{ - switch (access_size) - { - case 8: - { - MMIOReadCodeGenerator gen(this, registers_in_use, reg_value, address, sign_extend); - mmio->GetHandlerForRead(address).Visit(gen); - break; - } - case 16: - { - MMIOReadCodeGenerator gen(this, registers_in_use, reg_value, address, sign_extend); - mmio->GetHandlerForRead(address).Visit(gen); - break; - } - case 32: - { - MMIOReadCodeGenerator gen(this, registers_in_use, reg_value, address, sign_extend); - mmio->GetHandlerForRead(address).Visit(gen); - break; - } - } -} - -FixupBranch EmuCodeBlock::CheckIfSafeAddress(const OpArg& reg_value, X64Reg reg_addr, - BitSet32 registers_in_use) -{ - registers_in_use[reg_addr] = true; - if (reg_value.IsSimpleReg()) - registers_in_use[reg_value.GetSimpleReg()] = true; - - // Get ourselves a free register; try to pick one that doesn't involve pushing, if we can. - X64Reg scratch = RSCRATCH; - if (!registers_in_use[RSCRATCH]) - scratch = RSCRATCH; - else if (!registers_in_use[RSCRATCH_EXTRA]) - scratch = RSCRATCH_EXTRA; - else - scratch = reg_addr; - - if (scratch == reg_addr) - PUSH(scratch); - else - MOV(32, R(scratch), R(reg_addr)); - - // Perform lookup to see if we can use fast path. - SHR(32, R(scratch), Imm8(PowerPC::BAT_INDEX_SHIFT)); - TEST(32, MScaled(scratch, SCALE_4, PtrOffset(&PowerPC::dbat_table[0])), Imm32(2)); - - if (scratch == reg_addr) - POP(scratch); - - return J_CC(CC_Z, farcode.Enabled()); -} - -void EmuCodeBlock::SafeLoadToReg(X64Reg reg_value, const Gen::OpArg& opAddress, int accessSize, - s32 offset, BitSet32 registersInUse, bool signExtend, int flags) -{ - bool slowmem = (flags & SAFE_LOADSTORE_FORCE_SLOWMEM) != 0 || jit->jo.alwaysUseMemFuncs; - - registersInUse[reg_value] = false; - if (jit->jo.fastmem && !(flags & SAFE_LOADSTORE_NO_FASTMEM) && !slowmem) - { - u8* backpatchStart = GetWritableCodePtr(); - MovInfo mov; - bool offsetAddedToAddress = - UnsafeLoadToReg(reg_value, opAddress, accessSize, offset, signExtend, &mov); - TrampolineInfo& info = backPatchInfo[mov.address]; - info.pc = jit->js.compilerPC; - info.nonAtomicSwapStoreSrc = mov.nonAtomicSwapStore ? mov.nonAtomicSwapStoreSrc : INVALID_REG; - info.start = backpatchStart; - info.read = true; - info.op_reg = reg_value; - info.op_arg = opAddress; - info.offsetAddedToAddress = offsetAddedToAddress; - info.accessSize = accessSize >> 3; - info.offset = offset; - info.registersInUse = registersInUse; - info.flags = flags; - info.signExtend = signExtend; - ptrdiff_t padding = BACKPATCH_SIZE - (GetCodePtr() - backpatchStart); - if (padding > 0) - { - NOP(padding); - } - info.len = static_cast(GetCodePtr() - info.start); - - jit->js.fastmemLoadStore = mov.address; - return; - } - - if (opAddress.IsImm()) - { - u32 address = opAddress.Imm32() + offset; - SafeLoadToRegImmediate(reg_value, address, accessSize, registersInUse, signExtend); - return; - } - - _assert_msg_(DYNA_REC, opAddress.IsSimpleReg(), - "Incorrect use of SafeLoadToReg (address isn't register or immediate)"); - X64Reg reg_addr = opAddress.GetSimpleReg(); - if (offset) - { - reg_addr = RSCRATCH; - LEA(32, RSCRATCH, MDisp(opAddress.GetSimpleReg(), offset)); - } - - FixupBranch exit; - bool dr_set = (flags & SAFE_LOADSTORE_DR_ON) || UReg_MSR(MSR).DR; - bool fast_check_address = !slowmem && dr_set; - if (fast_check_address) - { - FixupBranch slow = CheckIfSafeAddress(R(reg_value), reg_addr, registersInUse); - UnsafeLoadToReg(reg_value, R(reg_addr), accessSize, 0, signExtend); - if (farcode.Enabled()) - SwitchToFarCode(); - else - exit = J(true); - SetJumpTarget(slow); - } - size_t rsp_alignment = (flags & SAFE_LOADSTORE_NO_PROLOG) ? 8 : 0; - ABI_PushRegistersAndAdjustStack(registersInUse, rsp_alignment); - switch (accessSize) - { - case 64: - ABI_CallFunctionR(PowerPC::Read_U64, reg_addr); - break; - case 32: - ABI_CallFunctionR(PowerPC::Read_U32, reg_addr); - break; - case 16: - ABI_CallFunctionR(PowerPC::Read_U16_ZX, reg_addr); - break; - case 8: - ABI_CallFunctionR(PowerPC::Read_U8_ZX, reg_addr); - break; - } - ABI_PopRegistersAndAdjustStack(registersInUse, rsp_alignment); - - MemoryExceptionCheck(); - if (signExtend && accessSize < 32) - { - // Need to sign extend values coming from the Read_U* functions. - MOVSX(32, accessSize, reg_value, R(ABI_RETURN)); - } - else if (reg_value != ABI_RETURN) - { - MOVZX(64, accessSize, reg_value, R(ABI_RETURN)); - } - - if (fast_check_address) - { - if (farcode.Enabled()) - { - exit = J(true); - SwitchToNearCode(); - } - SetJumpTarget(exit); - } -} - -void EmuCodeBlock::SafeLoadToRegImmediate(X64Reg reg_value, u32 address, int accessSize, - BitSet32 registersInUse, bool signExtend) -{ - // If the address is known to be RAM, just load it directly. - if (PowerPC::IsOptimizableRAMAddress(address)) - { - UnsafeLoadToReg(reg_value, Imm32(address), accessSize, 0, signExtend); - return; - } - - // If the address maps to an MMIO register, inline MMIO read code. - u32 mmioAddress = PowerPC::IsOptimizableMMIOAccess(address, accessSize); - if (accessSize != 64 && mmioAddress) - { - MMIOLoadToReg(Memory::mmio_mapping.get(), reg_value, registersInUse, mmioAddress, accessSize, - signExtend); - return; - } - - // Fall back to general-case code. - ABI_PushRegistersAndAdjustStack(registersInUse, 0); - switch (accessSize) - { - case 64: - ABI_CallFunctionC(PowerPC::Read_U64, address); - break; - case 32: - ABI_CallFunctionC(PowerPC::Read_U32, address); - break; - case 16: - ABI_CallFunctionC(PowerPC::Read_U16_ZX, address); - break; - case 8: - ABI_CallFunctionC(PowerPC::Read_U8_ZX, address); - break; - } - ABI_PopRegistersAndAdjustStack(registersInUse, 0); - - MemoryExceptionCheck(); - if (signExtend && accessSize < 32) - { - // Need to sign extend values coming from the Read_U* functions. - MOVSX(32, accessSize, reg_value, R(ABI_RETURN)); - } - else if (reg_value != ABI_RETURN) - { - MOVZX(64, accessSize, reg_value, R(ABI_RETURN)); - } -} - -static OpArg SwapImmediate(int accessSize, const OpArg& reg_value) -{ - if (accessSize == 32) - return Imm32(Common::swap32(reg_value.Imm32())); - else if (accessSize == 16) - return Imm16(Common::swap16(reg_value.Imm16())); - else - return Imm8(reg_value.Imm8()); -} - -void EmuCodeBlock::UnsafeWriteRegToReg(OpArg reg_value, X64Reg reg_addr, int accessSize, s32 offset, - bool swap, MovInfo* info) -{ - if (info) - { - info->address = GetWritableCodePtr(); - info->nonAtomicSwapStore = false; - } - - OpArg dest = MComplex(RMEM, reg_addr, SCALE_1, offset); - if (reg_value.IsImm()) - { - if (swap) - reg_value = SwapImmediate(accessSize, reg_value); - MOV(accessSize, dest, reg_value); - } - else if (swap) - { - SwapAndStore(accessSize, dest, reg_value.GetSimpleReg(), info); - } - else - { - MOV(accessSize, dest, reg_value); - } -} - -static OpArg FixImmediate(int accessSize, OpArg arg) -{ - if (arg.IsImm()) - { - arg = accessSize == 8 ? arg.AsImm8() : accessSize == 16 ? arg.AsImm16() : arg.AsImm32(); - } - return arg; -} - -void EmuCodeBlock::UnsafeWriteGatherPipe(int accessSize) -{ - // No need to protect these, they don't touch any state - // question - should we inline them instead? Pro: Lose a CALL Con: Code bloat - switch (accessSize) - { - case 8: - CALL(jit->GetAsmRoutines()->fifoDirectWrite8); - break; - case 16: - CALL(jit->GetAsmRoutines()->fifoDirectWrite16); - break; - case 32: - CALL(jit->GetAsmRoutines()->fifoDirectWrite32); - break; - case 64: - CALL(jit->GetAsmRoutines()->fifoDirectWrite64); - break; - } - jit->js.fifoBytesSinceCheck += accessSize >> 3; -} - -bool EmuCodeBlock::WriteToConstAddress(int accessSize, OpArg arg, u32 address, - BitSet32 registersInUse) -{ - arg = FixImmediate(accessSize, arg); - - // If we already know the address through constant folding, we can do some - // fun tricks... - if (jit->jo.optimizeGatherPipe && PowerPC::IsOptimizableGatherPipeWrite(address)) - { - if (!arg.IsSimpleReg(RSCRATCH)) - MOV(accessSize, R(RSCRATCH), arg); - - UnsafeWriteGatherPipe(accessSize); - return false; - } - else if (PowerPC::IsOptimizableRAMAddress(address)) - { - WriteToConstRamAddress(accessSize, arg, address); - return false; - } - else - { - // Helps external systems know which instruction triggered the write - MOV(32, PPCSTATE(pc), Imm32(jit->js.compilerPC)); - - ABI_PushRegistersAndAdjustStack(registersInUse, 0); - switch (accessSize) - { - case 64: - ABI_CallFunctionAC(64, PowerPC::Write_U64, arg, address); - break; - case 32: - ABI_CallFunctionAC(32, PowerPC::Write_U32, arg, address); - break; - case 16: - ABI_CallFunctionAC(16, PowerPC::Write_U16, arg, address); - break; - case 8: - ABI_CallFunctionAC(8, PowerPC::Write_U8, arg, address); - break; - } - ABI_PopRegistersAndAdjustStack(registersInUse, 0); - return true; - } -} - -void EmuCodeBlock::SafeWriteRegToReg(OpArg reg_value, X64Reg reg_addr, int accessSize, s32 offset, - BitSet32 registersInUse, int flags) -{ - bool swap = !(flags & SAFE_LOADSTORE_NO_SWAP); - bool slowmem = (flags & SAFE_LOADSTORE_FORCE_SLOWMEM) != 0 || jit->jo.alwaysUseMemFuncs; - - // set the correct immediate format - reg_value = FixImmediate(accessSize, reg_value); - - if (jit->jo.fastmem && !(flags & SAFE_LOADSTORE_NO_FASTMEM) && !slowmem) - { - u8* backpatchStart = GetWritableCodePtr(); - MovInfo mov; - UnsafeWriteRegToReg(reg_value, reg_addr, accessSize, offset, swap, &mov); - TrampolineInfo& info = backPatchInfo[mov.address]; - info.pc = jit->js.compilerPC; - info.nonAtomicSwapStoreSrc = mov.nonAtomicSwapStore ? mov.nonAtomicSwapStoreSrc : INVALID_REG; - info.start = backpatchStart; - info.read = false; - info.op_arg = reg_value; - info.op_reg = reg_addr; - info.offsetAddedToAddress = false; - info.accessSize = accessSize >> 3; - info.offset = offset; - info.registersInUse = registersInUse; - info.flags = flags; - ptrdiff_t padding = BACKPATCH_SIZE - (GetCodePtr() - backpatchStart); - if (padding > 0) - { - NOP(padding); - } - info.len = static_cast(GetCodePtr() - info.start); - - jit->js.fastmemLoadStore = mov.address; - - return; - } - - if (offset) - { - if (flags & SAFE_LOADSTORE_CLOBBER_RSCRATCH_INSTEAD_OF_ADDR) - { - LEA(32, RSCRATCH, MDisp(reg_addr, (u32)offset)); - reg_addr = RSCRATCH; - } - else - { - ADD(32, R(reg_addr), Imm32((u32)offset)); - } - } - - FixupBranch exit; - bool dr_set = (flags & SAFE_LOADSTORE_DR_ON) || UReg_MSR(MSR).DR; - bool fast_check_address = !slowmem && dr_set; - if (fast_check_address) - { - FixupBranch slow = CheckIfSafeAddress(reg_value, reg_addr, registersInUse); - UnsafeWriteRegToReg(reg_value, reg_addr, accessSize, 0, swap); - if (farcode.Enabled()) - SwitchToFarCode(); - else - exit = J(true); - SetJumpTarget(slow); - } - - // PC is used by memory watchpoints (if enabled) or to print accurate PC locations in debug logs - MOV(32, PPCSTATE(pc), Imm32(jit->js.compilerPC)); - - size_t rsp_alignment = (flags & SAFE_LOADSTORE_NO_PROLOG) ? 8 : 0; - ABI_PushRegistersAndAdjustStack(registersInUse, rsp_alignment); - - // If the input is an immediate, we need to put it in a register. - X64Reg reg; - if (reg_value.IsImm()) - { - reg = reg_addr == ABI_PARAM1 ? RSCRATCH : ABI_PARAM1; - MOV(accessSize, R(reg), reg_value); - } - else - { - reg = reg_value.GetSimpleReg(); - } - - switch (accessSize) - { - case 64: - ABI_CallFunctionRR(swap ? PowerPC::Write_U64 : PowerPC::Write_U64_Swap, reg, reg_addr); - break; - case 32: - ABI_CallFunctionRR(swap ? PowerPC::Write_U32 : PowerPC::Write_U32_Swap, reg, reg_addr); - break; - case 16: - ABI_CallFunctionRR(swap ? PowerPC::Write_U16 : PowerPC::Write_U16_Swap, reg, reg_addr); - break; - case 8: - ABI_CallFunctionRR(PowerPC::Write_U8, reg, reg_addr); - break; - } - ABI_PopRegistersAndAdjustStack(registersInUse, rsp_alignment); - - MemoryExceptionCheck(); - - if (fast_check_address) - { - if (farcode.Enabled()) - { - exit = J(true); - SwitchToNearCode(); - } - SetJumpTarget(exit); - } -} - -void EmuCodeBlock::WriteToConstRamAddress(int accessSize, OpArg arg, u32 address, bool swap) -{ - X64Reg reg; - if (arg.IsImm()) - { - arg = SwapImmediate(accessSize, arg); - MOV(32, R(RSCRATCH), Imm32(address)); - MOV(accessSize, MRegSum(RMEM, RSCRATCH), arg); - return; - } - - if (!arg.IsSimpleReg() || (!cpu_info.bMOVBE && swap && arg.GetSimpleReg() != RSCRATCH)) - { - MOV(accessSize, R(RSCRATCH), arg); - reg = RSCRATCH; - } - else - { - reg = arg.GetSimpleReg(); - } - - MOV(32, R(RSCRATCH2), Imm32(address)); - if (swap) - SwapAndStore(accessSize, MRegSum(RMEM, RSCRATCH2), reg); - else - MOV(accessSize, MRegSum(RMEM, RSCRATCH2), R(reg)); -} - -void EmuCodeBlock::ForceSinglePrecision(X64Reg output, const OpArg& input, bool packed, - bool duplicate) -{ - // Most games don't need these. Zelda requires it though - some platforms get stuck without them. - if (jit->jo.accurateSinglePrecision) - { - if (packed) - { - CVTPD2PS(output, input); - CVTPS2PD(output, R(output)); - } - else - { - CVTSD2SS(output, input); - CVTSS2SD(output, R(output)); - if (duplicate) - MOVDDUP(output, R(output)); - } - } - else if (!input.IsSimpleReg(output)) - { - if (duplicate) - MOVDDUP(output, input); - else - MOVAPD(output, input); - } -} - -// Abstract between AVX and SSE: automatically handle 3-operand instructions -void EmuCodeBlock::avx_op(void (XEmitter::*avxOp)(X64Reg, X64Reg, const OpArg&), - void (XEmitter::*sseOp)(X64Reg, const OpArg&), X64Reg regOp, - const OpArg& arg1, const OpArg& arg2, bool packed, bool reversible) -{ - if (arg1.IsSimpleReg(regOp)) - { - (this->*sseOp)(regOp, arg2); - } - else if (arg1.IsSimpleReg() && cpu_info.bAVX) - { - (this->*avxOp)(regOp, arg1.GetSimpleReg(), arg2); - } - else if (arg2.IsSimpleReg(regOp)) - { - if (reversible) - { - (this->*sseOp)(regOp, arg1); - } - else - { - // The ugly case: regOp == arg2 without AVX, or with arg1 == memory - if (!arg1.IsSimpleReg(XMM0)) - MOVAPD(XMM0, arg1); - if (cpu_info.bAVX) - { - (this->*avxOp)(regOp, XMM0, arg2); - } - else - { - (this->*sseOp)(XMM0, arg2); - if (packed) - MOVAPD(regOp, R(XMM0)); - else - MOVSD(regOp, R(XMM0)); - } - } - } - else - { - if (packed) - MOVAPD(regOp, arg1); - else - MOVSD(regOp, arg1); - (this->*sseOp)(regOp, arg1 == arg2 ? R(regOp) : arg2); - } -} - -// Abstract between AVX and SSE: automatically handle 3-operand instructions -void EmuCodeBlock::avx_op(void (XEmitter::*avxOp)(X64Reg, X64Reg, const OpArg&, u8), - void (XEmitter::*sseOp)(X64Reg, const OpArg&, u8), X64Reg regOp, - const OpArg& arg1, const OpArg& arg2, u8 imm) -{ - if (arg1.IsSimpleReg(regOp)) - { - (this->*sseOp)(regOp, arg2, imm); - } - else if (arg1.IsSimpleReg() && cpu_info.bAVX) - { - (this->*avxOp)(regOp, arg1.GetSimpleReg(), arg2, imm); - } - else if (arg2.IsSimpleReg(regOp)) - { - // The ugly case: regOp == arg2 without AVX, or with arg1 == memory - if (!arg1.IsSimpleReg(XMM0)) - MOVAPD(XMM0, arg1); - if (cpu_info.bAVX) - { - (this->*avxOp)(regOp, XMM0, arg2, imm); - } - else - { - (this->*sseOp)(XMM0, arg2, imm); - MOVAPD(regOp, R(XMM0)); - } - } - else - { - MOVAPD(regOp, arg1); - (this->*sseOp)(regOp, arg1 == arg2 ? R(regOp) : arg2, imm); - } -} - -alignas(16) static const u64 psMantissaTruncate[2] = {0xFFFFFFFFF8000000ULL, 0xFFFFFFFFF8000000ULL}; -alignas(16) static const u64 psRoundBit[2] = {0x8000000, 0x8000000}; - -// Emulate the odd truncation/rounding that the PowerPC does on the RHS operand before -// a single precision multiply. To be precise, it drops the low 28 bits of the mantissa, -// rounding to nearest as it does. -// It needs a temp, so let the caller pass that in. -void EmuCodeBlock::Force25BitPrecision(X64Reg output, const OpArg& input, X64Reg tmp) -{ - if (jit->jo.accurateSinglePrecision) - { - // mantissa = (mantissa & ~0xFFFFFFF) + ((mantissa & (1ULL << 27)) << 1); - if (input.IsSimpleReg() && cpu_info.bAVX) - { - VPAND(tmp, input.GetSimpleReg(), M(psRoundBit)); - VPAND(output, input.GetSimpleReg(), M(psMantissaTruncate)); - PADDQ(output, R(tmp)); - } - else - { - if (!input.IsSimpleReg(output)) - MOVAPD(output, input); - avx_op(&XEmitter::VPAND, &XEmitter::PAND, tmp, R(output), M(psRoundBit), true, true); - PAND(output, M(psMantissaTruncate)); - PADDQ(output, R(tmp)); - } - } - else if (!input.IsSimpleReg(output)) - { - MOVAPD(output, input); - } -} - -alignas(16) static u32 temp32; -alignas(16) static u64 temp64; - -// Since the following float conversion functions are used in non-arithmetic PPC float instructions, -// they must convert floats bitexact and never flush denormals to zero or turn SNaNs into QNaNs. -// This means we can't use CVTSS2SD/CVTSD2SS :( -// The x87 FPU doesn't even support flush-to-zero so we can use FLD+FSTP even on denormals. -// If the number is a NaN, make sure to set the QNaN bit back to its original value. - -// Another problem is that officially, converting doubles to single format results in undefined -// behavior. -// Relying on undefined behavior is a bug so no software should ever do this. -// In case it does happen, phire's more accurate implementation of ConvertDoubleToSingle() is -// reproduced below. - -//#define MORE_ACCURATE_DOUBLETOSINGLE -#ifdef MORE_ACCURATE_DOUBLETOSINGLE - -alignas(16) static const __m128i double_exponent = _mm_set_epi64x(0, 0x7ff0000000000000); -alignas(16) static const __m128i double_fraction = _mm_set_epi64x(0, 0x000fffffffffffff); -alignas(16) static const __m128i double_sign_bit = _mm_set_epi64x(0, 0x8000000000000000); -alignas(16) static const __m128i double_explicit_top_bit = _mm_set_epi64x(0, 0x0010000000000000); -alignas(16) static const __m128i double_top_two_bits = _mm_set_epi64x(0, 0xc000000000000000); -alignas(16) static const __m128i double_bottom_bits = _mm_set_epi64x(0, 0x07ffffffe0000000); - -// This is the same algorithm used in the interpreter (and actual hardware) -// The documentation states that the conversion of a double with an outside the -// valid range for a single (or a single denormal) is undefined. -// But testing on actual hardware shows it always picks bits 0..1 and 5..34 -// unless the exponent is in the range of 874 to 896. -void EmuCodeBlock::ConvertDoubleToSingle(X64Reg dst, X64Reg src) -{ - MOVSD(XMM1, R(src)); - - // Grab Exponent - PAND(XMM1, M(&double_exponent)); - PSRLQ(XMM1, 52); - MOVD_xmm(R(RSCRATCH), XMM1); - - // Check if the double is in the range of valid single subnormal - CMP(16, R(RSCRATCH), Imm16(896)); - FixupBranch NoDenormalize = J_CC(CC_G); - CMP(16, R(RSCRATCH), Imm16(874)); - FixupBranch NoDenormalize2 = J_CC(CC_L); - - // Denormalise - - // shift = (905 - Exponent) plus the 21 bit double to single shift - MOV(16, R(RSCRATCH), Imm16(905 + 21)); - MOVD_xmm(XMM0, R(RSCRATCH)); - PSUBQ(XMM0, R(XMM1)); - - // xmm1 = fraction | 0x0010000000000000 - MOVSD(XMM1, R(src)); - PAND(XMM1, M(&double_fraction)); - POR(XMM1, M(&double_explicit_top_bit)); - - // fraction >> shift - PSRLQ(XMM1, R(XMM0)); - - // OR the sign bit in. - MOVSD(XMM0, R(src)); - PAND(XMM0, M(&double_sign_bit)); - PSRLQ(XMM0, 32); - POR(XMM1, R(XMM0)); - - FixupBranch end = J(false); // Goto end - - SetJumpTarget(NoDenormalize); - SetJumpTarget(NoDenormalize2); - - // Don't Denormalize - - // We want bits 0, 1 - MOVSD(XMM1, R(src)); - PAND(XMM1, M(&double_top_two_bits)); - PSRLQ(XMM1, 32); - - // And 5 through to 34 - MOVSD(XMM0, R(src)); - PAND(XMM0, M(&double_bottom_bits)); - PSRLQ(XMM0, 29); - - // OR them togther - POR(XMM1, R(XMM0)); - - // End - SetJumpTarget(end); - MOVDDUP(dst, R(XMM1)); -} - -#else // MORE_ACCURATE_DOUBLETOSINGLE - -alignas(16) static const __m128i double_sign_bit = _mm_set_epi64x(0xffffffffffffffff, - 0x7fffffffffffffff); -alignas(16) static const __m128i single_qnan_bit = _mm_set_epi64x(0xffffffffffffffff, - 0xffffffffffbfffff); -alignas(16) static const __m128i double_qnan_bit = _mm_set_epi64x(0xffffffffffffffff, - 0xfff7ffffffffffff); - -// Smallest positive double that results in a normalized single. -alignas(16) static const double min_norm_single = std::numeric_limits::min(); - -void EmuCodeBlock::ConvertDoubleToSingle(X64Reg dst, X64Reg src) -{ - // Most games have flush-to-zero enabled, which causes the single -> double -> single process here - // to be lossy. - // This is a problem when games use float operations to copy non-float data. - // Changing the FPU mode is very expensive, so we can't do that. - // Here, check to see if the source is small enough that it will result in a denormal, and pass it - // to the x87 unit - // if it is. - avx_op(&XEmitter::VPAND, &XEmitter::PAND, XMM0, R(src), M(&double_sign_bit), true, true); - UCOMISD(XMM0, M(&min_norm_single)); - FixupBranch nanConversion = J_CC(CC_P, true); - FixupBranch denormalConversion = J_CC(CC_B, true); - CVTSD2SS(dst, R(src)); - - SwitchToFarCode(); - SetJumpTarget(nanConversion); - MOVQ_xmm(R(RSCRATCH), src); - // Put the quiet bit into CF. - BT(64, R(RSCRATCH), Imm8(51)); - CVTSD2SS(dst, R(src)); - FixupBranch continue1 = J_CC(CC_C, true); - // Clear the quiet bit of the SNaN, which was 0 (signalling) but got set to 1 (quiet) by - // conversion. - ANDPS(dst, M(&single_qnan_bit)); - FixupBranch continue2 = J(true); - - SetJumpTarget(denormalConversion); - MOVSD(M(&temp64), src); - FLD(64, M(&temp64)); - FSTP(32, M(&temp32)); - MOVSS(dst, M(&temp32)); - FixupBranch continue3 = J(true); - SwitchToNearCode(); - - SetJumpTarget(continue1); - SetJumpTarget(continue2); - SetJumpTarget(continue3); - // We'd normally need to MOVDDUP here to put the single in the top half of the output register - // too, but - // this function is only used to go directly to a following store, so we omit the MOVDDUP here. -} -#endif // MORE_ACCURATE_DOUBLETOSINGLE - -// Converting single->double is a bit easier because all single denormals are double normals. -void EmuCodeBlock::ConvertSingleToDouble(X64Reg dst, X64Reg src, bool src_is_gpr) -{ - X64Reg gprsrc = src_is_gpr ? src : RSCRATCH; - if (src_is_gpr) - { - MOVD_xmm(dst, R(src)); - } - else - { - if (dst != src) - MOVAPS(dst, R(src)); - MOVD_xmm(R(RSCRATCH), src); - } - - UCOMISS(dst, R(dst)); - CVTSS2SD(dst, R(dst)); - FixupBranch nanConversion = J_CC(CC_P, true); - - SwitchToFarCode(); - SetJumpTarget(nanConversion); - TEST(32, R(gprsrc), Imm32(0x00400000)); - FixupBranch continue1 = J_CC(CC_NZ, true); - ANDPD(dst, M(&double_qnan_bit)); - FixupBranch continue2 = J(true); - SwitchToNearCode(); - - SetJumpTarget(continue1); - SetJumpTarget(continue2); - MOVDDUP(dst, R(dst)); -} - -alignas(16) static const u64 psDoubleExp[2] = {0x7FF0000000000000ULL, 0}; -alignas(16) static const u64 psDoubleFrac[2] = {0x000FFFFFFFFFFFFFULL, 0}; -alignas(16) static const u64 psDoubleNoSign[2] = {0x7FFFFFFFFFFFFFFFULL, 0}; - -// TODO: it might be faster to handle FPRF in the same way as CR is currently handled for integer, -// storing -// the result of each floating point op and calculating it when needed. This is trickier than for -// integers -// though, because there's 32 possible FPRF bit combinations but only 9 categories of floating point -// values, -// which makes the whole thing rather trickier. -// Fortunately, PPCAnalyzer can optimize out a large portion of FPRF calculations, so maybe this -// isn't -// quite that necessary. -void EmuCodeBlock::SetFPRF(Gen::X64Reg xmm) -{ - AND(32, PPCSTATE(fpscr), Imm32(~FPRF_MASK)); - - FixupBranch continue1, continue2, continue3, continue4; - if (cpu_info.bSSE4_1) - { - MOVQ_xmm(R(RSCRATCH), xmm); - SHR(64, R(RSCRATCH), Imm8(63)); // Get the sign bit; almost all the branches need it. - PTEST(xmm, M(psDoubleExp)); - FixupBranch maxExponent = J_CC(CC_C); - FixupBranch zeroExponent = J_CC(CC_Z); - - // Nice normalized number: sign ? PPC_FPCLASS_NN : PPC_FPCLASS_PN; - LEA(32, RSCRATCH, MScaled(RSCRATCH, MathUtil::PPC_FPCLASS_NN - MathUtil::PPC_FPCLASS_PN, - MathUtil::PPC_FPCLASS_PN)); - continue1 = J(); - - SetJumpTarget(maxExponent); - PTEST(xmm, M(psDoubleFrac)); - FixupBranch notNAN = J_CC(CC_Z); - - // Max exponent + mantissa: PPC_FPCLASS_QNAN - MOV(32, R(RSCRATCH), Imm32(MathUtil::PPC_FPCLASS_QNAN)); - continue2 = J(); - - // Max exponent + no mantissa: sign ? PPC_FPCLASS_NINF : PPC_FPCLASS_PINF; - SetJumpTarget(notNAN); - LEA(32, RSCRATCH, MScaled(RSCRATCH, MathUtil::PPC_FPCLASS_NINF - MathUtil::PPC_FPCLASS_PINF, - MathUtil::PPC_FPCLASS_PINF)); - continue3 = J(); - - SetJumpTarget(zeroExponent); - PTEST(xmm, R(xmm)); - FixupBranch zero = J_CC(CC_Z); - - // No exponent + mantissa: sign ? PPC_FPCLASS_ND : PPC_FPCLASS_PD; - LEA(32, RSCRATCH, MScaled(RSCRATCH, MathUtil::PPC_FPCLASS_ND - MathUtil::PPC_FPCLASS_PD, - MathUtil::PPC_FPCLASS_PD)); - continue4 = J(); - - // Zero: sign ? PPC_FPCLASS_NZ : PPC_FPCLASS_PZ; - SetJumpTarget(zero); - SHL(32, R(RSCRATCH), Imm8(4)); - ADD(32, R(RSCRATCH), Imm8(MathUtil::PPC_FPCLASS_PZ)); - } - else - { - MOVQ_xmm(R(RSCRATCH), xmm); - TEST(64, R(RSCRATCH), M(psDoubleExp)); - FixupBranch zeroExponent = J_CC(CC_Z); - AND(64, R(RSCRATCH), M(psDoubleNoSign)); - CMP(64, R(RSCRATCH), M(psDoubleExp)); - FixupBranch nan = - J_CC(CC_G); // This works because if the sign bit is set, RSCRATCH is negative - FixupBranch infinity = J_CC(CC_E); - MOVQ_xmm(R(RSCRATCH), xmm); - SHR(64, R(RSCRATCH), Imm8(63)); - LEA(32, RSCRATCH, MScaled(RSCRATCH, MathUtil::PPC_FPCLASS_NN - MathUtil::PPC_FPCLASS_PN, - MathUtil::PPC_FPCLASS_PN)); - continue1 = J(); - SetJumpTarget(nan); - MOVQ_xmm(R(RSCRATCH), xmm); - SHR(64, R(RSCRATCH), Imm8(63)); - MOV(32, R(RSCRATCH), Imm32(MathUtil::PPC_FPCLASS_QNAN)); - continue2 = J(); - SetJumpTarget(infinity); - MOVQ_xmm(R(RSCRATCH), xmm); - SHR(64, R(RSCRATCH), Imm8(63)); - LEA(32, RSCRATCH, MScaled(RSCRATCH, MathUtil::PPC_FPCLASS_NINF - MathUtil::PPC_FPCLASS_PINF, - MathUtil::PPC_FPCLASS_PINF)); - continue3 = J(); - SetJumpTarget(zeroExponent); - TEST(64, R(RSCRATCH), R(RSCRATCH)); - FixupBranch zero = J_CC(CC_Z); - SHR(64, R(RSCRATCH), Imm8(63)); - LEA(32, RSCRATCH, MScaled(RSCRATCH, MathUtil::PPC_FPCLASS_ND - MathUtil::PPC_FPCLASS_PD, - MathUtil::PPC_FPCLASS_PD)); - continue4 = J(); - SetJumpTarget(zero); - SHR(64, R(RSCRATCH), Imm8(63)); - SHL(32, R(RSCRATCH), Imm8(4)); - ADD(32, R(RSCRATCH), Imm8(MathUtil::PPC_FPCLASS_PZ)); - } - - SetJumpTarget(continue1); - SetJumpTarget(continue2); - SetJumpTarget(continue3); - SetJumpTarget(continue4); - SHL(32, R(RSCRATCH), Imm8(FPRF_SHIFT)); - OR(32, PPCSTATE(fpscr), R(RSCRATCH)); -} - -void EmuCodeBlock::JitGetAndClearCAOV(bool oe) -{ - if (oe) - AND(8, PPCSTATE(xer_so_ov), Imm8(~XER_OV_MASK)); // XER.OV = 0 - SHR(8, PPCSTATE(xer_ca), Imm8(1)); // carry = XER.CA, XER.CA = 0 -} - -void EmuCodeBlock::JitSetCA() -{ - MOV(8, PPCSTATE(xer_ca), Imm8(1)); // XER.CA = 1 -} - -// Some testing shows CA is set roughly ~1/3 of the time (relative to clears), so -// branchless calculation of CA is probably faster in general. -void EmuCodeBlock::JitSetCAIf(CCFlags conditionCode) -{ - SETcc(conditionCode, PPCSTATE(xer_ca)); -} - -void EmuCodeBlock::JitClearCA() -{ - MOV(8, PPCSTATE(xer_ca), Imm8(0)); -} - -void EmuCodeBlock::Clear() -{ - backPatchInfo.clear(); - exceptionHandlerAtLoc.clear(); -} diff --git a/Source/Core/Core/PowerPC/JitCommon/Jit_Util.h b/Source/Core/Core/PowerPC/JitCommon/Jit_Util.h deleted file mode 100644 index 2be1233cc4..0000000000 --- a/Source/Core/Core/PowerPC/JitCommon/Jit_Util.h +++ /dev/null @@ -1,211 +0,0 @@ -// Copyright 2010 Dolphin Emulator Project -// Licensed under GPLv2+ -// Refer to the license.txt file included./ - -#pragma once - -#include - -#include "Common/BitSet.h" -#include "Common/CPUDetect.h" -#include "Common/x64Emitter.h" -#include "Core/PowerPC/PowerPC.h" - -namespace MMIO -{ -class Mapping; -} - -// We offset by 0x80 because the range of one byte memory offsets is -// -0x80..0x7f. -#define PPCSTATE(x) \ - MDisp(RPPCSTATE, (int)((char*)&PowerPC::ppcState.x - (char*)&PowerPC::ppcState) - 0x80) -// In case you want to disable the ppcstate register: -// #define PPCSTATE(x) M(&PowerPC::ppcState.x) -#define PPCSTATE_LR PPCSTATE(spr[SPR_LR]) -#define PPCSTATE_CTR PPCSTATE(spr[SPR_CTR]) -#define PPCSTATE_SRR0 PPCSTATE(spr[SPR_SRR0]) -#define PPCSTATE_SRR1 PPCSTATE(spr[SPR_SRR1]) - -// A place to throw blocks of code we don't want polluting the cache, e.g. rarely taken -// exception branches. -class FarCodeCache : public Gen::X64CodeBlock -{ -private: - bool m_enabled = false; - -public: - bool Enabled() const { return m_enabled; } - void Init(int size) - { - AllocCodeSpace(size); - m_enabled = true; - } - void Shutdown() - { - FreeCodeSpace(); - m_enabled = false; - } -}; - -static const int CODE_SIZE = 1024 * 1024 * 32; - -// a bit of a hack; the MMU results in a vast amount more code ending up in the far cache, -// mostly exception handling, so give it a whole bunch more space if the MMU is on. -static const int FARCODE_SIZE = 1024 * 1024 * 8; -static const int FARCODE_SIZE_MMU = 1024 * 1024 * 48; - -// same for the trampoline code cache, because fastmem results in far more backpatches in MMU mode -static const int TRAMPOLINE_CODE_SIZE = 1024 * 1024 * 8; -static const int TRAMPOLINE_CODE_SIZE_MMU = 1024 * 1024 * 32; - -// Stores information we need to batch-patch a MOV with a call to the slow read/write path after -// it faults. There will be 10s of thousands of these structs live, so be wary of making this too -// big. -struct TrampolineInfo final -{ - // The start of the store operation that failed -- we will patch a JMP here - u8* start; - - // The start + len = end of the store operation (points to the next instruction) - u32 len; - - // The PPC PC for the current load/store block - u32 pc; - - // Saved because we need these to make the ABI call in the trampoline - BitSet32 registersInUse; - - // The MOV operation - Gen::X64Reg nonAtomicSwapStoreSrc; - - // src/dest for load/store - s32 offset; - Gen::X64Reg op_reg; - Gen::OpArg op_arg; - - // Original SafeLoadXXX/SafeStoreXXX flags - u8 flags; - - // Memory access size (in bytes) - u8 accessSize : 4; - - // true if this is a read op vs a write - bool read : 1; - - // for read operations, true if needs sign-extension after load - bool signExtend : 1; - - // Set to true if we added the offset to the address and need to undo it - bool offsetAddedToAddress : 1; -}; - -// Like XCodeBlock but has some utilities for memory access. -class EmuCodeBlock : public Gen::X64CodeBlock -{ -public: - FarCodeCache farcode; - u8* nearcode; // Backed up when we switch to far code. - - void MemoryExceptionCheck(); - - // Simple functions to switch between near and far code emitting - void SwitchToFarCode() - { - nearcode = GetWritableCodePtr(); - SetCodePtr(farcode.GetWritableCodePtr()); - } - - void SwitchToNearCode() - { - farcode.SetCodePtr(GetWritableCodePtr()); - SetCodePtr(nearcode); - } - - Gen::FixupBranch CheckIfSafeAddress(const Gen::OpArg& reg_value, Gen::X64Reg reg_addr, - BitSet32 registers_in_use); - void UnsafeLoadRegToReg(Gen::X64Reg reg_addr, Gen::X64Reg reg_value, int accessSize, - s32 offset = 0, bool signExtend = false); - void UnsafeLoadRegToRegNoSwap(Gen::X64Reg reg_addr, Gen::X64Reg reg_value, int accessSize, - s32 offset, bool signExtend = false); - // these return the address of the MOV, for backpatching - void UnsafeWriteRegToReg(Gen::OpArg reg_value, Gen::X64Reg reg_addr, int accessSize, - s32 offset = 0, bool swap = true, Gen::MovInfo* info = nullptr); - void UnsafeWriteRegToReg(Gen::X64Reg reg_value, Gen::X64Reg reg_addr, int accessSize, - s32 offset = 0, bool swap = true, Gen::MovInfo* info = nullptr) - { - UnsafeWriteRegToReg(R(reg_value), reg_addr, accessSize, offset, swap, info); - } - bool UnsafeLoadToReg(Gen::X64Reg reg_value, Gen::OpArg opAddress, int accessSize, s32 offset, - bool signExtend, Gen::MovInfo* info = nullptr); - void UnsafeWriteGatherPipe(int accessSize); - - // Generate a load/write from the MMIO handler for a given address. Only - // call for known addresses in MMIO range (MMIO::IsMMIOAddress). - void MMIOLoadToReg(MMIO::Mapping* mmio, Gen::X64Reg reg_value, BitSet32 registers_in_use, - u32 address, int access_size, bool sign_extend); - - enum SafeLoadStoreFlags - { - SAFE_LOADSTORE_NO_SWAP = 1, - SAFE_LOADSTORE_NO_PROLOG = 2, - // This indicates that the write being generated cannot be patched (and thus can't use fastmem) - SAFE_LOADSTORE_NO_FASTMEM = 4, - SAFE_LOADSTORE_CLOBBER_RSCRATCH_INSTEAD_OF_ADDR = 8, - // Force slowmem (used when generating fallbacks in trampolines) - SAFE_LOADSTORE_FORCE_SLOWMEM = 16, - SAFE_LOADSTORE_DR_ON = 32, - }; - - void SafeLoadToReg(Gen::X64Reg reg_value, const Gen::OpArg& opAddress, int accessSize, s32 offset, - BitSet32 registersInUse, bool signExtend, int flags = 0); - void SafeLoadToRegImmediate(Gen::X64Reg reg_value, u32 address, int accessSize, - BitSet32 registersInUse, bool signExtend); - - // Clobbers RSCRATCH or reg_addr depending on the relevant flag. Preserves - // reg_value if the load fails and js.memcheck is enabled. - // Works with immediate inputs and simple registers only. - void SafeWriteRegToReg(Gen::OpArg reg_value, Gen::X64Reg reg_addr, int accessSize, s32 offset, - BitSet32 registersInUse, int flags = 0); - void SafeWriteRegToReg(Gen::X64Reg reg_value, Gen::X64Reg reg_addr, int accessSize, s32 offset, - BitSet32 registersInUse, int flags = 0) - { - SafeWriteRegToReg(R(reg_value), reg_addr, accessSize, offset, registersInUse, flags); - } - - // applies to safe and unsafe WriteRegToReg - bool WriteClobbersRegValue(int accessSize, bool swap) - { - return swap && !cpu_info.bMOVBE && accessSize > 8; - } - - void WriteToConstRamAddress(int accessSize, Gen::OpArg arg, u32 address, bool swap = true); - // returns true if an exception could have been caused - bool WriteToConstAddress(int accessSize, Gen::OpArg arg, u32 address, BitSet32 registersInUse); - void JitGetAndClearCAOV(bool oe); - void JitSetCA(); - void JitSetCAIf(Gen::CCFlags conditionCode); - void JitClearCA(); - - void avx_op(void (Gen::XEmitter::*avxOp)(Gen::X64Reg, Gen::X64Reg, const Gen::OpArg&), - void (Gen::XEmitter::*sseOp)(Gen::X64Reg, const Gen::OpArg&), Gen::X64Reg regOp, - const Gen::OpArg& arg1, const Gen::OpArg& arg2, bool packed = true, - bool reversible = false); - void avx_op(void (Gen::XEmitter::*avxOp)(Gen::X64Reg, Gen::X64Reg, const Gen::OpArg&, u8), - void (Gen::XEmitter::*sseOp)(Gen::X64Reg, const Gen::OpArg&, u8), Gen::X64Reg regOp, - const Gen::OpArg& arg1, const Gen::OpArg& arg2, u8 imm); - - void ForceSinglePrecision(Gen::X64Reg output, const Gen::OpArg& input, bool packed = true, - bool duplicate = false); - void Force25BitPrecision(Gen::X64Reg output, const Gen::OpArg& input, Gen::X64Reg tmp); - - // RSCRATCH might get trashed - void ConvertSingleToDouble(Gen::X64Reg dst, Gen::X64Reg src, bool src_is_gpr = false); - void ConvertDoubleToSingle(Gen::X64Reg dst, Gen::X64Reg src); - void SetFPRF(Gen::X64Reg xmm); - void Clear(); - -protected: - std::unordered_map backPatchInfo; - std::unordered_map exceptionHandlerAtLoc; -}; diff --git a/Source/Core/Core/PowerPC/JitCommon/TrampolineCache.cpp b/Source/Core/Core/PowerPC/JitCommon/TrampolineCache.cpp deleted file mode 100644 index 79c0a5abee..0000000000 --- a/Source/Core/Core/PowerPC/JitCommon/TrampolineCache.cpp +++ /dev/null @@ -1,85 +0,0 @@ -// Copyright 2014 Dolphin Emulator Project -// Licensed under GPLv2+ -// Refer to the license.txt file included. - -#include -#include - -#include "Common/CommonFuncs.h" -#include "Common/CommonTypes.h" -#include "Common/JitRegister.h" -#include "Common/x64ABI.h" -#include "Common/x64Emitter.h" -#include "Core/PowerPC/JitCommon/JitBase.h" -#include "Core/PowerPC/JitCommon/Jit_Util.h" -#include "Core/PowerPC/JitCommon/TrampolineCache.h" -#include "Core/PowerPC/PowerPC.h" - -#ifdef _WIN32 -#include -#endif - -using namespace Gen; - -void TrampolineCache::Init(int size) -{ - AllocCodeSpace(size); -} - -void TrampolineCache::ClearCodeSpace() -{ - X64CodeBlock::ClearCodeSpace(); -} - -void TrampolineCache::Shutdown() -{ - FreeCodeSpace(); -} - -const u8* TrampolineCache::GenerateTrampoline(const TrampolineInfo& info) -{ - if (info.read) - { - return GenerateReadTrampoline(info); - } - - return GenerateWriteTrampoline(info); -} - -const u8* TrampolineCache::GenerateReadTrampoline(const TrampolineInfo& info) -{ - if (GetSpaceLeft() < 1024) - PanicAlert("Trampoline cache full"); - - const u8* trampoline = GetCodePtr(); - - SafeLoadToReg(info.op_reg, info.op_arg, info.accessSize << 3, info.offset, info.registersInUse, - info.signExtend, info.flags | SAFE_LOADSTORE_FORCE_SLOWMEM); - - JMP(info.start + info.len, true); - - JitRegister::Register(trampoline, GetCodePtr(), "JIT_ReadTrampoline_%x", info.pc); - return trampoline; -} - -const u8* TrampolineCache::GenerateWriteTrampoline(const TrampolineInfo& info) -{ - if (GetSpaceLeft() < 1024) - PanicAlert("Trampoline cache full"); - - const u8* trampoline = GetCodePtr(); - - // Don't treat FIFO writes specially for now because they require a burst - // check anyway. - - // PC is used by memory watchpoints (if enabled) or to print accurate PC locations in debug logs - MOV(32, PPCSTATE(pc), Imm32(info.pc)); - - SafeWriteRegToReg(info.op_arg, info.op_reg, info.accessSize << 3, info.offset, - info.registersInUse, info.flags | SAFE_LOADSTORE_FORCE_SLOWMEM); - - JMP(info.start + info.len, true); - - JitRegister::Register(trampoline, GetCodePtr(), "JIT_WriteTrampoline_%x", info.pc); - return trampoline; -} diff --git a/Source/Core/Core/PowerPC/JitCommon/TrampolineCache.h b/Source/Core/Core/PowerPC/JitCommon/TrampolineCache.h deleted file mode 100644 index c43668dc8d..0000000000 --- a/Source/Core/Core/PowerPC/JitCommon/TrampolineCache.h +++ /dev/null @@ -1,27 +0,0 @@ -// Copyright 2014 Dolphin Emulator Project -// Licensed under GPLv2+ -// Refer to the license.txt file included. - -#pragma once - -#include "Common/BitSet.h" -#include "Common/CommonTypes.h" -#include "Common/x64Emitter.h" -#include "Core/PowerPC/JitCommon/Jit_Util.h" - -struct InstructionInfo; - -// We need at least this many bytes for backpatching. -const int BACKPATCH_SIZE = 5; - -class TrampolineCache : public EmuCodeBlock -{ - const u8* GenerateReadTrampoline(const TrampolineInfo& info); - const u8* GenerateWriteTrampoline(const TrampolineInfo& info); - -public: - void Init(int size); - void Shutdown(); - const u8* GenerateTrampoline(const TrampolineInfo& info); - void ClearCodeSpace(); -}; diff --git a/Source/Core/Core/PowerPC/JitILCommon/JitILBase.h b/Source/Core/Core/PowerPC/JitILCommon/JitILBase.h index fec6631a40..228ccd269d 100644 --- a/Source/Core/Core/PowerPC/JitILCommon/JitILBase.h +++ b/Source/Core/Core/PowerPC/JitILCommon/JitILBase.h @@ -6,7 +6,7 @@ #include "Common/CommonTypes.h" #include "Core/PowerPC/Gekko.h" -#include "Core/PowerPC/JitCommon/JitBase.h" +#include "Core/PowerPC/Jit64Common/Jit64Base.h" #include "Core/PowerPC/JitILCommon/IR.h" #include "Core/PowerPC/PPCAnalyst.h" -- cgit v1.2.3