IOS: checkpoint Starlet JIT, PPC cache and Wii timing fixes
Save the remaining ARM JIT and MMU/cache optimizations, accurate Starlet timer and Wiimote report cadence, opt-in PPC event tracing, and full texture hashing in the LLE launcher. Include regression coverage and exclude local profiling artifacts. Validation: 127 targeted tests from 15 suites passed, with one disabled test. Includes the current user-tested source state following the persistent NAND milestone.
This commit is contained in:
@@ -66,3 +66,8 @@ CMakeUserPresets.json
|
|||||||
/capstone_local/
|
/capstone_local/
|
||||||
/capstone_runtime/
|
/capstone_runtime/
|
||||||
/capstone-*.whl
|
/capstone-*.whl
|
||||||
|
|
||||||
|
# Local profiling helpers, captures, and diagnostic output
|
||||||
|
/.codex-tools/
|
||||||
|
/.starlet-profile/
|
||||||
|
/yaya48-starlet-log.txt
|
||||||
|
|||||||
+13
-2
@@ -5,6 +5,8 @@ param(
|
|||||||
|
|
||||||
[switch]$DisableJIT,
|
[switch]$DisableJIT,
|
||||||
|
|
||||||
|
[switch]$SafeTextureCache = $true,
|
||||||
|
|
||||||
[switch]$Wait
|
[switch]$Wait
|
||||||
)
|
)
|
||||||
|
|
||||||
@@ -16,7 +18,7 @@ if (-not (Test-Path -LiteralPath $dolphinPath)) {
|
|||||||
throw "DolphinNoGUI.exe is missing: $dolphinPath"
|
throw "DolphinNoGUI.exe is missing: $dolphinPath"
|
||||||
}
|
}
|
||||||
|
|
||||||
$dolphin = Start-Process -FilePath $dolphinPath -ArgumentList @(
|
$dolphinArguments = @(
|
||||||
'-u', $userPath,
|
'-u', $userPath,
|
||||||
'-n', '0000000100000002',
|
'-n', '0000000100000002',
|
||||||
'-C', "Dolphin.Core.WiiStarletJIT=$(-not $DisableJIT)",
|
'-C', "Dolphin.Core.WiiStarletJIT=$(-not $DisableJIT)",
|
||||||
@@ -27,7 +29,16 @@ $dolphin = Start-Process -FilePath $dolphinPath -ArgumentList @(
|
|||||||
'-C', 'Logger.Logs.IOS=True',
|
'-C', 'Logger.Logs.IOS=True',
|
||||||
'-v', 'D3D',
|
'-v', 'D3D',
|
||||||
'-p', 'win32'
|
'-p', 'win32'
|
||||||
) -WorkingDirectory $repoRoot -PassThru
|
)
|
||||||
|
|
||||||
|
if ($SafeTextureCache) {
|
||||||
|
# Hash every texture byte: sparse samples can miss small CPU-rendered text updates.
|
||||||
|
# This is a per-run override and does not change the saved graphics configuration.
|
||||||
|
$dolphinArguments += @('-C', 'Graphics.Settings.SafeTextureCacheColorSamples=0')
|
||||||
|
}
|
||||||
|
|
||||||
|
$dolphin = Start-Process -FilePath $dolphinPath -ArgumentList $dolphinArguments `
|
||||||
|
-WorkingDirectory $repoRoot -PassThru
|
||||||
|
|
||||||
$dolphin.PriorityClass = 'High'
|
$dolphin.PriorityClass = 'High'
|
||||||
|
|
||||||
|
|||||||
@@ -317,7 +317,7 @@ void VideoInterfaceManager::RegisterMMIO(MMIO::Mapping* mmio, u32 base)
|
|||||||
mmio->Register(
|
mmio->Register(
|
||||||
base | VI_VERTICAL_BEAM_POSITION, MMIO::ComplexRead<u16>([](Core::System& system, u32) {
|
base | VI_VERTICAL_BEAM_POSITION, MMIO::ComplexRead<u16>([](Core::System& system, u32) {
|
||||||
auto& vi = system.GetVideoInterface();
|
auto& vi = system.GetVideoInterface();
|
||||||
return 1 + (vi.m_half_line_count) / 2;
|
return vi.GetVerticalBeamPosition();
|
||||||
}),
|
}),
|
||||||
MMIO::ComplexWrite<u16>([](Core::System& system, u32, u16 val) {
|
MMIO::ComplexWrite<u16>([](Core::System& system, u32, u16 val) {
|
||||||
auto& vi = system.GetVideoInterface();
|
auto& vi = system.GetVideoInterface();
|
||||||
|
|||||||
@@ -389,6 +389,9 @@ public:
|
|||||||
u32 GetTicksPerHalfLine() const;
|
u32 GetTicksPerHalfLine() const;
|
||||||
u32 GetTicksPerField() const;
|
u32 GetTicksPerField() const;
|
||||||
|
|
||||||
|
// Current one-based vertical beam position exposed by VI_VERTICAL_BEAM_POSITION.
|
||||||
|
u16 GetVerticalBeamPosition() const { return static_cast<u16>(1 + m_half_line_count / 2); }
|
||||||
|
|
||||||
// Not adjusted by VBI Clock Override.
|
// Not adjusted by VBI Clock Override.
|
||||||
u32 GetNominalTicksPerHalfLine() const;
|
u32 GetNominalTicksPerHalfLine() const;
|
||||||
|
|
||||||
|
|||||||
@@ -7,6 +7,7 @@
|
|||||||
#include <array>
|
#include <array>
|
||||||
#include <bit>
|
#include <bit>
|
||||||
#include <cassert>
|
#include <cassert>
|
||||||
|
#include <cstring>
|
||||||
#include <limits>
|
#include <limits>
|
||||||
#include <utility>
|
#include <utility>
|
||||||
|
|
||||||
@@ -201,43 +202,95 @@ u32 ARMCore::TranslateVirtualAddress(u32 address) const
|
|||||||
return physical_address;
|
return physical_address;
|
||||||
};
|
};
|
||||||
|
|
||||||
|
return cache_translation(WalkPageTables(modified_address, nullptr, nullptr, nullptr, nullptr));
|
||||||
|
}
|
||||||
|
|
||||||
|
u32 ARMCore::TranslateVirtualAddressForJit(u32 address, const u8** first_descriptor,
|
||||||
|
u32* first_descriptor_value,
|
||||||
|
const u8** second_descriptor,
|
||||||
|
u32* second_descriptor_value) const
|
||||||
|
{
|
||||||
|
*first_descriptor = nullptr;
|
||||||
|
*first_descriptor_value = 0;
|
||||||
|
*second_descriptor = nullptr;
|
||||||
|
*second_descriptor_value = 0;
|
||||||
|
if ((m_cp15.control & CP15_CONTROL_MMU) == 0)
|
||||||
|
return address;
|
||||||
|
|
||||||
|
const u32 modified_address =
|
||||||
|
address < 0x02000000 ? address | (m_cp15.process_id & 0xfe000000) : address;
|
||||||
|
const u32 physical_address =
|
||||||
|
WalkPageTables(modified_address, first_descriptor, first_descriptor_value, second_descriptor,
|
||||||
|
second_descriptor_value);
|
||||||
|
|
||||||
|
// Keep the interpreter and generated data-access path coherent with the translation established
|
||||||
|
// for this native code block.
|
||||||
|
const u32 virtual_page = modified_address >> 10;
|
||||||
|
TLBEntry& entry = m_tlb[GetTLBIndex(virtual_page)];
|
||||||
|
entry.virtual_page = virtual_page;
|
||||||
|
entry.physical_page = physical_address & ~0x3ffU;
|
||||||
|
entry.generation = m_tlb_generation;
|
||||||
|
return physical_address;
|
||||||
|
}
|
||||||
|
|
||||||
|
u32 ARMCore::WalkPageTables(u32 modified_address, const u8** first_descriptor,
|
||||||
|
u32* first_descriptor_value, const u8** second_descriptor,
|
||||||
|
u32* second_descriptor_value) const
|
||||||
|
{
|
||||||
|
const auto read_descriptor = [&](u32 physical_address, const u8** host_pointer,
|
||||||
|
u32* host_value) {
|
||||||
|
const u8* const pointer = m_bus.GetDirectMemoryPointer(physical_address, sizeof(u32));
|
||||||
|
if (host_pointer)
|
||||||
|
*host_pointer = pointer;
|
||||||
|
if (host_value)
|
||||||
|
{
|
||||||
|
*host_value = 0;
|
||||||
|
if (pointer)
|
||||||
|
std::memcpy(host_value, pointer, sizeof(u32));
|
||||||
|
}
|
||||||
|
return ReadPhysical32(physical_address);
|
||||||
|
};
|
||||||
|
|
||||||
const u32 first_level_address =
|
const u32 first_level_address =
|
||||||
(m_cp15.translation_table_base & 0xffffc000) | ((modified_address >> 18) & 0x3ffc);
|
(m_cp15.translation_table_base & 0xffffc000) | ((modified_address >> 18) & 0x3ffc);
|
||||||
const u32 first_level = ReadPhysical32(first_level_address);
|
const u32 first_level =
|
||||||
|
read_descriptor(first_level_address, first_descriptor, first_descriptor_value);
|
||||||
switch (first_level & 3)
|
switch (first_level & 3)
|
||||||
{
|
{
|
||||||
case 1: // Coarse second-level table.
|
case 1: // Coarse second-level table.
|
||||||
{
|
{
|
||||||
const u32 second_level_address =
|
const u32 second_level_address =
|
||||||
(first_level & 0xfffffc00) | ((modified_address >> 10) & 0x3fc);
|
(first_level & 0xfffffc00) | ((modified_address >> 10) & 0x3fc);
|
||||||
const u32 second_level = ReadPhysical32(second_level_address);
|
const u32 second_level =
|
||||||
|
read_descriptor(second_level_address, second_descriptor, second_descriptor_value);
|
||||||
switch (second_level & 3)
|
switch (second_level & 3)
|
||||||
{
|
{
|
||||||
case 1: // 64 KiB large page.
|
case 1: // 64 KiB large page.
|
||||||
return cache_translation((second_level & 0xffff0000) | (modified_address & 0xffff));
|
return (second_level & 0xffff0000) | (modified_address & 0xffff);
|
||||||
case 2:
|
case 2:
|
||||||
case 3: // 4 KiB small page; extended small pages share this mapping shape.
|
case 3: // 4 KiB small page; extended small pages share this mapping shape.
|
||||||
return cache_translation((second_level & 0xfffff000) | (modified_address & 0xfff));
|
return (second_level & 0xfffff000) | (modified_address & 0xfff);
|
||||||
default:
|
default:
|
||||||
return cache_translation(modified_address);
|
return modified_address;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
case 2: // 1 MiB section.
|
case 2: // 1 MiB section.
|
||||||
return cache_translation((first_level & 0xfff00000) | (modified_address & 0x000fffff));
|
return (first_level & 0xfff00000) | (modified_address & 0x000fffff);
|
||||||
case 3: // Fine second-level table.
|
case 3: // Fine second-level table.
|
||||||
{
|
{
|
||||||
const u32 second_level_address = (first_level & 0xfffff000) | ((modified_address >> 8) & 0xffc);
|
const u32 second_level_address = (first_level & 0xfffff000) | ((modified_address >> 8) & 0xffc);
|
||||||
const u32 second_level = ReadPhysical32(second_level_address);
|
const u32 second_level =
|
||||||
|
read_descriptor(second_level_address, second_descriptor, second_descriptor_value);
|
||||||
switch (second_level & 3)
|
switch (second_level & 3)
|
||||||
{
|
{
|
||||||
case 1:
|
case 1:
|
||||||
return cache_translation((second_level & 0xffff0000) | (modified_address & 0xffff));
|
return (second_level & 0xffff0000) | (modified_address & 0xffff);
|
||||||
case 2:
|
case 2:
|
||||||
return cache_translation((second_level & 0xfffff000) | (modified_address & 0xfff));
|
return (second_level & 0xfffff000) | (modified_address & 0xfff);
|
||||||
case 3: // 1 KiB tiny page.
|
case 3: // 1 KiB tiny page.
|
||||||
return cache_translation((second_level & 0xfffffc00) | (modified_address & 0x3ff));
|
return (second_level & 0xfffffc00) | (modified_address & 0x3ff);
|
||||||
default:
|
default:
|
||||||
return cache_translation(modified_address);
|
return modified_address;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
default:
|
default:
|
||||||
@@ -246,7 +299,7 @@ u32 ARMCore::TranslateVirtualAddress(u32 address) const
|
|||||||
// fallback keeps diagnostics observable meanwhile. Cache that provisional translation just
|
// fallback keeps diagnostics observable meanwhile. Cache that provisional translation just
|
||||||
// like a mapped page: real software must invalidate the TLB after changing a page-table entry,
|
// like a mapped page: real software must invalidate the TLB after changing a page-table entry,
|
||||||
// and without this cache the JIT would side-exit forever on every identity access.
|
// and without this cache the JIT would side-exit forever on every identity access.
|
||||||
return cache_translation(modified_address);
|
return modified_address;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -864,6 +917,213 @@ size_t ARMCore::GetJitCompiledBlockCount() const
|
|||||||
#endif
|
#endif
|
||||||
}
|
}
|
||||||
|
|
||||||
|
u64 ARMCore::GetJitLifetimeCompiledBlockCount() const
|
||||||
|
{
|
||||||
|
#if defined(_M_X86_64)
|
||||||
|
return m_jit ? m_jit->GetLifetimeCompiledBlockCount() : 0;
|
||||||
|
#else
|
||||||
|
return 0;
|
||||||
|
#endif
|
||||||
|
}
|
||||||
|
|
||||||
|
u64 ARMCore::GetJitBlockLookupCount() const
|
||||||
|
{
|
||||||
|
#if defined(_M_X86_64)
|
||||||
|
return m_jit ? m_jit->GetBlockLookupCount() : 0;
|
||||||
|
#else
|
||||||
|
return 0;
|
||||||
|
#endif
|
||||||
|
}
|
||||||
|
|
||||||
|
u64 ARMCore::GetJitBlockCacheHitCount() const
|
||||||
|
{
|
||||||
|
#if defined(_M_X86_64)
|
||||||
|
return m_jit ? m_jit->GetBlockCacheHitCount() : 0;
|
||||||
|
#else
|
||||||
|
return 0;
|
||||||
|
#endif
|
||||||
|
}
|
||||||
|
|
||||||
|
u64 ARMCore::GetJitDispatchSlowCount() const
|
||||||
|
{
|
||||||
|
#if defined(_M_X86_64)
|
||||||
|
return m_jit ? m_jit->GetDispatchSlowCount() : 0;
|
||||||
|
#else
|
||||||
|
return 0;
|
||||||
|
#endif
|
||||||
|
}
|
||||||
|
|
||||||
|
u64 ARMCore::GetJitDispatchStaleGenerationCount() const
|
||||||
|
{
|
||||||
|
#if defined(_M_X86_64)
|
||||||
|
return m_jit ? m_jit->GetDispatchStaleGenerationCount() : 0;
|
||||||
|
#else
|
||||||
|
return 0;
|
||||||
|
#endif
|
||||||
|
}
|
||||||
|
|
||||||
|
u64 ARMCore::GetJitDispatchPhysicalAliasCount() const
|
||||||
|
{
|
||||||
|
#if defined(_M_X86_64)
|
||||||
|
return m_jit ? m_jit->GetDispatchPhysicalAliasCount() : 0;
|
||||||
|
#else
|
||||||
|
return 0;
|
||||||
|
#endif
|
||||||
|
}
|
||||||
|
|
||||||
|
u64 ARMCore::GetJitDispatchCollisionCount() const
|
||||||
|
{
|
||||||
|
#if defined(_M_X86_64)
|
||||||
|
return m_jit ? m_jit->GetDispatchCollisionCount() : 0;
|
||||||
|
#else
|
||||||
|
return 0;
|
||||||
|
#endif
|
||||||
|
}
|
||||||
|
|
||||||
|
u64 ARMCore::GetJitDispatchKeyMissCount() const
|
||||||
|
{
|
||||||
|
#if defined(_M_X86_64)
|
||||||
|
return m_jit ? m_jit->GetDispatchKeyMissCount() : 0;
|
||||||
|
#else
|
||||||
|
return 0;
|
||||||
|
#endif
|
||||||
|
}
|
||||||
|
|
||||||
|
u64 ARMCore::GetJitDispatchPageMissCount() const
|
||||||
|
{
|
||||||
|
#if defined(_M_X86_64)
|
||||||
|
return m_jit ? m_jit->GetDispatchPageMissCount() : 0;
|
||||||
|
#else
|
||||||
|
return 0;
|
||||||
|
#endif
|
||||||
|
}
|
||||||
|
|
||||||
|
u64 ARMCore::GetJitDispatchEmptyEntryCount() const
|
||||||
|
{
|
||||||
|
#if defined(_M_X86_64)
|
||||||
|
return m_jit ? m_jit->GetDispatchEmptyEntryCount() : 0;
|
||||||
|
#else
|
||||||
|
return 0;
|
||||||
|
#endif
|
||||||
|
}
|
||||||
|
|
||||||
|
u64 ARMCore::GetJitDispatchGeneratedKeyMismatchCount() const
|
||||||
|
{
|
||||||
|
#if defined(_M_X86_64)
|
||||||
|
return m_jit ? m_jit->GetDispatchGeneratedKeyMismatchCount() : 0;
|
||||||
|
#else
|
||||||
|
return 0;
|
||||||
|
#endif
|
||||||
|
}
|
||||||
|
|
||||||
|
u64 ARMCore::GetJitDispatchGeneratedSetMismatchCount() const
|
||||||
|
{
|
||||||
|
#if defined(_M_X86_64)
|
||||||
|
return m_jit ? m_jit->GetDispatchGeneratedSetMismatchCount() : 0;
|
||||||
|
#else
|
||||||
|
return 0;
|
||||||
|
#endif
|
||||||
|
}
|
||||||
|
|
||||||
|
u64 ARMCore::GetJitDispatchKeyMissWithExactMatchCount() const
|
||||||
|
{
|
||||||
|
#if defined(_M_X86_64)
|
||||||
|
return m_jit ? m_jit->GetDispatchKeyMissWithExactMatchCount() : 0;
|
||||||
|
#else
|
||||||
|
return 0;
|
||||||
|
#endif
|
||||||
|
}
|
||||||
|
|
||||||
|
u64 ARMCore::GetJitDispatchKeyMissWithPhysicalAliasCount() const
|
||||||
|
{
|
||||||
|
#if defined(_M_X86_64)
|
||||||
|
return m_jit ? m_jit->GetDispatchKeyMissWithPhysicalAliasCount() : 0;
|
||||||
|
#else
|
||||||
|
return 0;
|
||||||
|
#endif
|
||||||
|
}
|
||||||
|
|
||||||
|
u64 ARMCore::GetJitDispatchKeyMissWithSamePhysicalPageCount() const
|
||||||
|
{
|
||||||
|
#if defined(_M_X86_64)
|
||||||
|
return m_jit ? m_jit->GetDispatchKeyMissWithSamePhysicalPageCount() : 0;
|
||||||
|
#else
|
||||||
|
return 0;
|
||||||
|
#endif
|
||||||
|
}
|
||||||
|
|
||||||
|
u64 ARMCore::GetJitDispatchKeyMissWithEmptySlotCount() const
|
||||||
|
{
|
||||||
|
#if defined(_M_X86_64)
|
||||||
|
return m_jit ? m_jit->GetDispatchKeyMissWithEmptySlotCount() : 0;
|
||||||
|
#else
|
||||||
|
return 0;
|
||||||
|
#endif
|
||||||
|
}
|
||||||
|
|
||||||
|
u64 ARMCore::GetJitDispatchKeyMissWithFullSetCount() const
|
||||||
|
{
|
||||||
|
#if defined(_M_X86_64)
|
||||||
|
return m_jit ? m_jit->GetDispatchKeyMissWithFullSetCount() : 0;
|
||||||
|
#else
|
||||||
|
return 0;
|
||||||
|
#endif
|
||||||
|
}
|
||||||
|
|
||||||
|
u64 ARMCore::GetJitDispatchKeyMissWithStaleTranslationCount() const
|
||||||
|
{
|
||||||
|
#if defined(_M_X86_64)
|
||||||
|
return m_jit ? m_jit->GetDispatchKeyMissWithStaleTranslationCount() : 0;
|
||||||
|
#else
|
||||||
|
return 0;
|
||||||
|
#endif
|
||||||
|
}
|
||||||
|
|
||||||
|
u64 ARMCore::GetJitFastEntryEmptyInsertCount() const
|
||||||
|
{
|
||||||
|
#if defined(_M_X86_64)
|
||||||
|
return m_jit ? m_jit->GetFastEntryEmptyInsertCount() : 0;
|
||||||
|
#else
|
||||||
|
return 0;
|
||||||
|
#endif
|
||||||
|
}
|
||||||
|
|
||||||
|
u64 ARMCore::GetJitFastEntryReplacementCount() const
|
||||||
|
{
|
||||||
|
#if defined(_M_X86_64)
|
||||||
|
return m_jit ? m_jit->GetFastEntryReplacementCount() : 0;
|
||||||
|
#else
|
||||||
|
return 0;
|
||||||
|
#endif
|
||||||
|
}
|
||||||
|
|
||||||
|
u64 ARMCore::GetJitInstructionCacheClearCount() const
|
||||||
|
{
|
||||||
|
#if defined(_M_X86_64)
|
||||||
|
return m_jit ? m_jit->GetInstructionCacheClearCount() : 0;
|
||||||
|
#else
|
||||||
|
return 0;
|
||||||
|
#endif
|
||||||
|
}
|
||||||
|
|
||||||
|
u64 ARMCore::GetJitCodeSpaceClearCount() const
|
||||||
|
{
|
||||||
|
#if defined(_M_X86_64)
|
||||||
|
return m_jit ? m_jit->GetCodeSpaceClearCount() : 0;
|
||||||
|
#else
|
||||||
|
return 0;
|
||||||
|
#endif
|
||||||
|
}
|
||||||
|
|
||||||
|
u64 ARMCore::GetJitDiscardedBlockCount() const
|
||||||
|
{
|
||||||
|
#if defined(_M_X86_64)
|
||||||
|
return m_jit ? m_jit->GetDiscardedBlockCount() : 0;
|
||||||
|
#else
|
||||||
|
return 0;
|
||||||
|
#endif
|
||||||
|
}
|
||||||
|
|
||||||
u64 ARMCore::GetJitAddressTranslationCount() const
|
u64 ARMCore::GetJitAddressTranslationCount() const
|
||||||
{
|
{
|
||||||
#if defined(_M_X86_64)
|
#if defined(_M_X86_64)
|
||||||
@@ -990,6 +1250,16 @@ u64 ARMCore::GetJitSlowSRAMPageWriteAccessCount(size_t page) const
|
|||||||
#endif
|
#endif
|
||||||
}
|
}
|
||||||
|
|
||||||
|
std::vector<std::pair<u64, u64>> ARMCore::TakeHotJitSlowMemorySamples(size_t maximum_count)
|
||||||
|
{
|
||||||
|
#if defined(_M_X86_64)
|
||||||
|
return m_jit ? m_jit->TakeHotSlowMemorySamples(maximum_count) :
|
||||||
|
std::vector<std::pair<u64, u64>>{};
|
||||||
|
#else
|
||||||
|
return {};
|
||||||
|
#endif
|
||||||
|
}
|
||||||
|
|
||||||
void ARMCore::RecordJitFallback(u32 address, bool thumb)
|
void ARMCore::RecordJitFallback(u32 address, bool thumb)
|
||||||
{
|
{
|
||||||
++m_jit_fallback_instruction_count;
|
++m_jit_fallback_instruction_count;
|
||||||
|
|||||||
@@ -64,6 +64,10 @@ public:
|
|||||||
virtual u8* GetFastmemSRAMBase() const { return nullptr; }
|
virtual u8* GetFastmemSRAMBase() const { return nullptr; }
|
||||||
virtual const bool* GetFastmemBoot0Mapped() const { return nullptr; }
|
virtual const bool* GetFastmemBoot0Mapped() const { return nullptr; }
|
||||||
virtual const bool* GetFastmemSRAMSplitMode() const { return nullptr; }
|
virtual const bool* GetFastmemSRAMSplitMode() const { return nullptr; }
|
||||||
|
// Returns a stable host pointer for ordinary physical memory. The pointer is used only to
|
||||||
|
// validate ARM page-table descriptors after guest TLB maintenance; MMIO and overlaid ROM must
|
||||||
|
// return nullptr so their observable reads continue through the bus.
|
||||||
|
virtual const u8* GetDirectMemoryPointer(u32 address, u32 size) const { return nullptr; }
|
||||||
};
|
};
|
||||||
|
|
||||||
class ARMCore final
|
class ARMCore final
|
||||||
@@ -164,6 +168,29 @@ public:
|
|||||||
u64 GetJitExecutedInstructions() const;
|
u64 GetJitExecutedInstructions() const;
|
||||||
u64 GetJitNativeExecutedInstructions() const;
|
u64 GetJitNativeExecutedInstructions() const;
|
||||||
size_t GetJitCompiledBlockCount() const;
|
size_t GetJitCompiledBlockCount() const;
|
||||||
|
u64 GetJitLifetimeCompiledBlockCount() const;
|
||||||
|
u64 GetJitBlockLookupCount() const;
|
||||||
|
u64 GetJitBlockCacheHitCount() const;
|
||||||
|
u64 GetJitDispatchSlowCount() const;
|
||||||
|
u64 GetJitDispatchStaleGenerationCount() const;
|
||||||
|
u64 GetJitDispatchPhysicalAliasCount() const;
|
||||||
|
u64 GetJitDispatchCollisionCount() const;
|
||||||
|
u64 GetJitDispatchKeyMissCount() const;
|
||||||
|
u64 GetJitDispatchPageMissCount() const;
|
||||||
|
u64 GetJitDispatchEmptyEntryCount() const;
|
||||||
|
u64 GetJitDispatchGeneratedKeyMismatchCount() const;
|
||||||
|
u64 GetJitDispatchGeneratedSetMismatchCount() const;
|
||||||
|
u64 GetJitDispatchKeyMissWithExactMatchCount() const;
|
||||||
|
u64 GetJitDispatchKeyMissWithPhysicalAliasCount() const;
|
||||||
|
u64 GetJitDispatchKeyMissWithSamePhysicalPageCount() const;
|
||||||
|
u64 GetJitDispatchKeyMissWithEmptySlotCount() const;
|
||||||
|
u64 GetJitDispatchKeyMissWithFullSetCount() const;
|
||||||
|
u64 GetJitDispatchKeyMissWithStaleTranslationCount() const;
|
||||||
|
u64 GetJitFastEntryEmptyInsertCount() const;
|
||||||
|
u64 GetJitFastEntryReplacementCount() const;
|
||||||
|
u64 GetJitInstructionCacheClearCount() const;
|
||||||
|
u64 GetJitCodeSpaceClearCount() const;
|
||||||
|
u64 GetJitDiscardedBlockCount() const;
|
||||||
u64 GetJitAddressTranslationCount() const;
|
u64 GetJitAddressTranslationCount() const;
|
||||||
u64 GetJitSlowReadCount() const;
|
u64 GetJitSlowReadCount() const;
|
||||||
u64 GetJitSlowWriteCount() const;
|
u64 GetJitSlowWriteCount() const;
|
||||||
@@ -178,6 +205,7 @@ public:
|
|||||||
u64 GetJitSlowSRAMPageAccessCount(size_t page) const;
|
u64 GetJitSlowSRAMPageAccessCount(size_t page) const;
|
||||||
u64 GetJitSlowSRAMPageReadAccessCount(size_t page) const;
|
u64 GetJitSlowSRAMPageReadAccessCount(size_t page) const;
|
||||||
u64 GetJitSlowSRAMPageWriteAccessCount(size_t page) const;
|
u64 GetJitSlowSRAMPageWriteAccessCount(size_t page) const;
|
||||||
|
std::vector<std::pair<u64, u64>> TakeHotJitSlowMemorySamples(size_t maximum_count);
|
||||||
u64 GetJitFallbackInstructionCount() const { return m_jit_fallback_instruction_count; }
|
u64 GetJitFallbackInstructionCount() const { return m_jit_fallback_instruction_count; }
|
||||||
u64 GetMemoryPollEntryCount() const { return m_memory_poll_entry_count; }
|
u64 GetMemoryPollEntryCount() const { return m_memory_poll_entry_count; }
|
||||||
std::vector<HotPCSample> GetHotPCSamples(size_t maximum_count);
|
std::vector<HotPCSample> GetHotPCSamples(size_t maximum_count);
|
||||||
@@ -246,6 +274,12 @@ private:
|
|||||||
void Write32(u32 address, u32 value);
|
void Write32(u32 address, u32 value);
|
||||||
void WriteByte(u32 address, u8 value);
|
void WriteByte(u32 address, u8 value);
|
||||||
u32 TranslateVirtualAddress(u32 address) const;
|
u32 TranslateVirtualAddress(u32 address) const;
|
||||||
|
u32 TranslateVirtualAddressForJit(u32 address, const u8** first_descriptor,
|
||||||
|
u32* first_descriptor_value, const u8** second_descriptor,
|
||||||
|
u32* second_descriptor_value) const;
|
||||||
|
u32 WalkPageTables(u32 modified_address, const u8** first_descriptor,
|
||||||
|
u32* first_descriptor_value, const u8** second_descriptor,
|
||||||
|
u32* second_descriptor_value) const;
|
||||||
u32 ReadPhysical32(u32 address) const;
|
u32 ReadPhysical32(u32 address) const;
|
||||||
void InvalidateTLB();
|
void InvalidateTLB();
|
||||||
u16 FetchThumbInstruction(u32 address);
|
u16 FetchThumbInstruction(u32 address);
|
||||||
|
|||||||
@@ -76,6 +76,10 @@ ARMJitX64::ARMJitX64(ARMCore& core) : m_core(core)
|
|||||||
m_executed_instructions_offset =
|
m_executed_instructions_offset =
|
||||||
static_cast<s32>(reinterpret_cast<const u8*>(&m_core.m_executed_instructions) - base);
|
static_cast<s32>(reinterpret_cast<const u8*>(&m_core.m_executed_instructions) - base);
|
||||||
m_control_offset = static_cast<s32>(reinterpret_cast<const u8*>(&m_core.m_cp15.control) - base);
|
m_control_offset = static_cast<s32>(reinterpret_cast<const u8*>(&m_core.m_cp15.control) - base);
|
||||||
|
m_translation_table_base_offset = static_cast<s32>(
|
||||||
|
reinterpret_cast<const u8*>(&m_core.m_cp15.translation_table_base) - base);
|
||||||
|
m_domain_access_control_offset = static_cast<s32>(
|
||||||
|
reinterpret_cast<const u8*>(&m_core.m_cp15.domain_access_control) - base);
|
||||||
m_process_id_offset =
|
m_process_id_offset =
|
||||||
static_cast<s32>(reinterpret_cast<const u8*>(&m_core.m_cp15.process_id) - base);
|
static_cast<s32>(reinterpret_cast<const u8*>(&m_core.m_cp15.process_id) - base);
|
||||||
m_tlb_generation_offset =
|
m_tlb_generation_offset =
|
||||||
@@ -99,6 +103,18 @@ void ARMJitX64::PoisonMemory()
|
|||||||
}
|
}
|
||||||
|
|
||||||
void ARMJitX64::Clear()
|
void ARMJitX64::Clear()
|
||||||
|
{
|
||||||
|
++m_instruction_cache_clear_count;
|
||||||
|
RequestClear();
|
||||||
|
}
|
||||||
|
|
||||||
|
void ARMJitX64::ClearForCodeSpace()
|
||||||
|
{
|
||||||
|
++m_code_space_clear_count;
|
||||||
|
RequestClear();
|
||||||
|
}
|
||||||
|
|
||||||
|
void ARMJitX64::RequestClear()
|
||||||
{
|
{
|
||||||
// CP15 I-cache maintenance can be reached through a fallback at the end of the currently
|
// CP15 I-cache maintenance can be reached through a fallback at the end of the currently
|
||||||
// executing host block. Defer destruction until its RET has brought us back to C++.
|
// executing host block. Defer destruction until its RET has brought us back to C++.
|
||||||
@@ -107,8 +123,15 @@ void ARMJitX64::Clear()
|
|||||||
m_clear_pending = true;
|
m_clear_pending = true;
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
|
ClearCodeCache();
|
||||||
|
}
|
||||||
|
|
||||||
|
void ARMJitX64::ClearCodeCache()
|
||||||
|
{
|
||||||
|
m_discarded_block_count += m_blocks.size();
|
||||||
m_blocks.clear();
|
m_blocks.clear();
|
||||||
std::ranges::fill(m_fast_entries, FastEntry{});
|
std::ranges::fill(m_fast_entries, FastEntry{});
|
||||||
|
std::ranges::fill(m_fast_entry_next_victim, 0);
|
||||||
SetCodePtr(m_block_code_begin, region + region_size);
|
SetCodePtr(m_block_code_begin, region + region_size);
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -117,13 +140,12 @@ void ARMJitX64::InvalidateTranslationContext()
|
|||||||
// A TLB invalidation changes which physical page a virtual PC resolves to, but it does not
|
// A TLB invalidation changes which physical page a virtual PC resolves to, but it does not
|
||||||
// invalidate the ARM instruction cache. Keep already translated physical code and force the
|
// invalidate the ARM instruction cache. Keep already translated physical code and force the
|
||||||
// generated dispatcher to resolve the next virtual PC through the current page tables. Every
|
// generated dispatcher to resolve the next virtual PC through the current page tables. Every
|
||||||
// fast entry carries the ARMCore TLB generation, so the common invalidation is O(1). This is
|
// shared page translation carries the ARMCore TLB generation, so the common invalidation is
|
||||||
// critical for IOS, which flushes its TLB on virtually every process switch; clearing the whole
|
// O(1). This is critical for IOS, which flushes its TLB on virtually every process switch.
|
||||||
// 65,536-entry array here previously consumed most of the host CPU. Generation zero is skipped by
|
// Generation zero is skipped by ARMCore. If the 32-bit counter eventually wraps back to one,
|
||||||
// ARMCore. If the 32-bit counter eventually wraps back to one, clear ancient generation-one
|
// clear ancient generation-one page translations once to prevent a generation alias.
|
||||||
// entries once to prevent an alias after the wrap.
|
|
||||||
if (m_core.m_tlb_generation == 1)
|
if (m_core.m_tlb_generation == 1)
|
||||||
std::ranges::fill(m_fast_entries, FastEntry{});
|
std::ranges::fill(m_fast_translations, FastTranslationEntry{});
|
||||||
}
|
}
|
||||||
|
|
||||||
u32 ARMJitX64::Run(u64 cycle_budget)
|
u32 ARMJitX64::Run(u64 cycle_budget)
|
||||||
@@ -138,7 +160,7 @@ u32 ARMJitX64::Run(u64 cycle_budget)
|
|||||||
if (m_clear_pending)
|
if (m_clear_pending)
|
||||||
{
|
{
|
||||||
m_clear_pending = false;
|
m_clear_pending = false;
|
||||||
Clear();
|
ClearCodeCache();
|
||||||
}
|
}
|
||||||
const u32 executed = static_cast<u32>(result);
|
const u32 executed = static_cast<u32>(result);
|
||||||
m_executed_instructions += executed;
|
m_executed_instructions += executed;
|
||||||
@@ -168,9 +190,11 @@ void ARMJitX64::GenerateDispatcher()
|
|||||||
CMP(8, MatR(RAX), Imm8(0));
|
CMP(8, MatR(RAX), Imm8(0));
|
||||||
FixupBranch invalidated = J_CC(CC_NE, Jump::Near);
|
FixupBranch invalidated = J_CC(CC_NE, Jump::Near);
|
||||||
|
|
||||||
// Direct-mapped native block cache. The key includes CPSR.T in bit zero and uses the ARM926
|
// Four-way native block cache. The key includes CPSR.T in bit zero and uses the ARM926 modified
|
||||||
// modified virtual address (MVA) for low FCSE addresses. Different IOS process identifiers can
|
// virtual address (MVA) for low FCSE addresses. Different IOS process identifiers can therefore
|
||||||
// therefore retain independent hot entries without invalidating the cache on every c13 write.
|
// retain independent hot entries without invalidating the cache on every c13 write. Four ways
|
||||||
|
// are important here: IOS regularly alternates among several hot basic blocks whose MVAs used
|
||||||
|
// to evict each other millions of times in the old direct-mapped cache.
|
||||||
MOV(32, R(EAX), MRegister(15));
|
MOV(32, R(EAX), MRegister(15));
|
||||||
MOV(32, R(ECX), MCPSR());
|
MOV(32, R(ECX), MCPSR());
|
||||||
SHR(32, R(ECX), Imm8(5));
|
SHR(32, R(ECX), Imm8(5));
|
||||||
@@ -187,30 +211,181 @@ void ARMJitX64::GenerateDispatcher()
|
|||||||
SetJumpTarget(mmu_disabled);
|
SetJumpTarget(mmu_disabled);
|
||||||
SetJumpTarget(outside_fcse);
|
SetJumpTarget(outside_fcse);
|
||||||
|
|
||||||
// Fold the FCSE PID bits down into the cache index. A plain low-bit mask makes every process
|
// Multiplicative hashing folds both FCSE PID and low instruction-address bits into the set. Each
|
||||||
// collide because PID occupies MVA[31:25]. The full MVA remains in FastEntry::key for safety.
|
// set occupies one 64-byte host cache line (four 16-byte FastEntry values).
|
||||||
MOV(32, R(EDX), R(EAX));
|
MOV(32, R(EDX), R(EAX));
|
||||||
SHR(32, R(EDX), Imm8(1));
|
IMUL(32, EDX, R(EDX), Imm32(0x9e3779b1U));
|
||||||
MOV(32, R(ECX), R(EAX));
|
SHR(32, R(EDX), Imm8(16));
|
||||||
SHR(32, R(ECX), Imm8(17));
|
SHL(64, R(RDX), Imm8(6));
|
||||||
XOR(32, R(EDX), R(ECX));
|
|
||||||
AND(32, R(EDX), Imm32(static_cast<u32>(FAST_ENTRY_COUNT - 1)));
|
|
||||||
SHL(64, R(RDX), Imm8(4));
|
|
||||||
MOV(64, R(R11), ImmPtr(m_fast_entries.data()));
|
MOV(64, R(R11), ImmPtr(m_fast_entries.data()));
|
||||||
ADD(64, R(R11), R(RDX));
|
ADD(64, R(R11), R(RDX));
|
||||||
MOV(32, R(R8), MDisp(JIT_CORE, m_tlb_generation_offset));
|
|
||||||
CMP(32, MDisp(R11, static_cast<s32>(offsetof(FastEntry, tlb_generation))), R(R8));
|
// With the MMU disabled the virtual and physical identities are identical, so a key-only lookup
|
||||||
FixupBranch stale_generation = J_CC(CC_NE);
|
// is sufficient and TLB maintenance is irrelevant.
|
||||||
|
MOV(32, R(ECX), MDisp(JIT_CORE, m_control_offset));
|
||||||
|
TEST(32, R(ECX), Imm32(1));
|
||||||
|
FixupBranch mmu_fast_translation = J_CC(CC_NE, Jump::Near);
|
||||||
|
MOV(32, R(R8), R(EAX));
|
||||||
|
AND(32, R(R8), Imm32(~0x3ffU));
|
||||||
CMP(32, MDisp(R11, static_cast<s32>(offsetof(FastEntry, key))), R(EAX));
|
CMP(32, MDisp(R11, static_cast<s32>(offsetof(FastEntry, key))), R(EAX));
|
||||||
FixupBranch cache_miss = J_CC(CC_NE);
|
FixupBranch no_mmu_way0_key_miss = J_CC(CC_NE, Jump::Near);
|
||||||
|
CMP(32, MDisp(R11, static_cast<s32>(offsetof(FastEntry, physical_page))), R(R8));
|
||||||
|
FixupBranch no_mmu_hit_way0 = J_CC(CC_E, Jump::Near);
|
||||||
|
SetJumpTarget(no_mmu_way0_key_miss);
|
||||||
|
CMP(32, MDisp(R11, static_cast<s32>(sizeof(FastEntry) + offsetof(FastEntry, key))), R(EAX));
|
||||||
|
FixupBranch no_mmu_way1_key_miss = J_CC(CC_NE, Jump::Near);
|
||||||
|
CMP(32, MDisp(R11, static_cast<s32>(sizeof(FastEntry) + offsetof(FastEntry, physical_page))),
|
||||||
|
R(R8));
|
||||||
|
FixupBranch no_mmu_hit_way1 = J_CC(CC_E, Jump::Near);
|
||||||
|
SetJumpTarget(no_mmu_way1_key_miss);
|
||||||
|
CMP(32, MDisp(R11, static_cast<s32>(2 * sizeof(FastEntry) + offsetof(FastEntry, key))), R(EAX));
|
||||||
|
FixupBranch no_mmu_way2_key_miss = J_CC(CC_NE, Jump::Near);
|
||||||
|
CMP(32, MDisp(R11,
|
||||||
|
static_cast<s32>(2 * sizeof(FastEntry) + offsetof(FastEntry, physical_page))),
|
||||||
|
R(R8));
|
||||||
|
FixupBranch no_mmu_hit_way2 = J_CC(CC_E, Jump::Near);
|
||||||
|
SetJumpTarget(no_mmu_way2_key_miss);
|
||||||
|
CMP(32, MDisp(R11, static_cast<s32>(3 * sizeof(FastEntry) + offsetof(FastEntry, key))), R(EAX));
|
||||||
|
FixupBranch no_mmu_cache_miss_key = J_CC(CC_NE, Jump::Near);
|
||||||
|
CMP(32, MDisp(R11,
|
||||||
|
static_cast<s32>(3 * sizeof(FastEntry) + offsetof(FastEntry, physical_page))),
|
||||||
|
R(R8));
|
||||||
|
FixupBranch no_mmu_cache_miss_page = J_CC(CC_NE, Jump::Near);
|
||||||
|
ADD(64, R(R11), Imm8(static_cast<u8>(3 * sizeof(FastEntry))));
|
||||||
|
FixupBranch no_mmu_way_selected = J(Jump::Near);
|
||||||
|
SetJumpTarget(no_mmu_hit_way2);
|
||||||
|
ADD(64, R(R11), Imm8(static_cast<u8>(2 * sizeof(FastEntry))));
|
||||||
|
FixupBranch no_mmu_way2_selected = J(Jump::Near);
|
||||||
|
SetJumpTarget(no_mmu_hit_way1);
|
||||||
|
ADD(64, R(R11), Imm8(static_cast<u8>(sizeof(FastEntry))));
|
||||||
|
SetJumpTarget(no_mmu_hit_way0);
|
||||||
|
SetJumpTarget(no_mmu_way2_selected);
|
||||||
|
SetJumpTarget(no_mmu_way_selected);
|
||||||
|
MOV(64, R(R10), MDisp(R11, static_cast<s32>(offsetof(FastEntry, entry))));
|
||||||
|
TEST(64, R(R10), R(R10));
|
||||||
|
FixupBranch empty_entry_no_mmu = J_CC(CC_Z, Jump::Near);
|
||||||
|
JMPptr(R(R10));
|
||||||
|
SetJumpTarget(mmu_fast_translation);
|
||||||
|
|
||||||
|
// ARM926 TLB maintenance invalidates translations by page. Validate one shared 1 KiB physical
|
||||||
|
// identity per page instead of sending every decoded block through C++ after each IOS context
|
||||||
|
// switch. A true page remap still rejects the old native block below.
|
||||||
|
MOV(32, R(R9), R(EAX));
|
||||||
|
SHR(32, R(R9), Imm8(10));
|
||||||
|
MOV(32, R(R8), R(R9));
|
||||||
|
MOV(32, R(ECX), R(R9));
|
||||||
|
SHR(32, R(ECX), Imm8(12));
|
||||||
|
XOR(32, R(R8), R(ECX));
|
||||||
|
AND(32, R(R8), Imm32(static_cast<u32>(FAST_TRANSLATION_ENTRY_COUNT - 1)));
|
||||||
|
SHL(64, R(R8), Imm8(6));
|
||||||
|
MOV(64, R(R10), ImmPtr(m_fast_translations.data()));
|
||||||
|
ADD(64, R(R10), R(R8));
|
||||||
|
|
||||||
|
CMP(32, MDisp(R10, static_cast<s32>(offsetof(FastTranslationEntry, virtual_page))), R(R9));
|
||||||
|
FixupBranch stale_page = J_CC(CC_NE, Jump::Near);
|
||||||
|
MOV(32, R(R8), MDisp(JIT_CORE, m_tlb_generation_offset));
|
||||||
|
CMP(32, MDisp(R10, static_cast<s32>(offsetof(FastTranslationEntry, tlb_generation))), R(R8));
|
||||||
|
FixupBranch current_translation = J_CC(CC_E, Jump::Near);
|
||||||
|
|
||||||
|
// IOS invalidates its ARM926 TLB on almost every process switch, even when the page tables are
|
||||||
|
// unchanged. Revalidate the exact descriptors that established this translation in generated
|
||||||
|
// code. A changed TTBR or descriptor still takes the complete C++ page-table walk below, while
|
||||||
|
// unchanged mappings avoid millions of dispatcher side exits.
|
||||||
|
MOV(32, R(ECX), MDisp(JIT_CORE, m_translation_table_base_offset));
|
||||||
|
CMP(32, MDisp(R10, static_cast<s32>(offsetof(FastTranslationEntry, translation_table_base))),
|
||||||
|
R(ECX));
|
||||||
|
FixupBranch stale_translation_table = J_CC(CC_NE, Jump::Near);
|
||||||
|
MOV(64, R(R8), MDisp(R10, static_cast<s32>(offsetof(FastTranslationEntry, first_descriptor))));
|
||||||
|
TEST(64, R(R8), R(R8));
|
||||||
|
FixupBranch unavailable_first_descriptor = J_CC(CC_Z, Jump::Near);
|
||||||
|
MOV(32, R(ECX), MatR(R8));
|
||||||
|
CMP(32, R(ECX),
|
||||||
|
MDisp(R10, static_cast<s32>(offsetof(FastTranslationEntry, first_descriptor_value))));
|
||||||
|
FixupBranch changed_first_descriptor = J_CC(CC_NE, Jump::Near);
|
||||||
|
MOV(64, R(R8), MDisp(R10, static_cast<s32>(offsetof(FastTranslationEntry, second_descriptor))));
|
||||||
|
TEST(64, R(R8), R(R8));
|
||||||
|
FixupBranch no_second_descriptor = J_CC(CC_Z, Jump::Near);
|
||||||
|
MOV(32, R(ECX), MatR(R8));
|
||||||
|
CMP(32, R(ECX),
|
||||||
|
MDisp(R10, static_cast<s32>(offsetof(FastTranslationEntry, second_descriptor_value))));
|
||||||
|
FixupBranch changed_second_descriptor = J_CC(CC_NE, Jump::Near);
|
||||||
|
SetJumpTarget(no_second_descriptor);
|
||||||
|
MOV(32, R(R8), MDisp(JIT_CORE, m_tlb_generation_offset));
|
||||||
|
MOV(32, MDisp(R10, static_cast<s32>(offsetof(FastTranslationEntry, tlb_generation))), R(R8));
|
||||||
|
SetJumpTarget(current_translation);
|
||||||
|
MOV(32, R(R8), MDisp(R10, static_cast<s32>(offsetof(FastTranslationEntry, physical_page))));
|
||||||
|
|
||||||
|
// The same MVA can legitimately resolve to different physical pages in different IOS address
|
||||||
|
// spaces. Select a way only when both identities match, allowing the four physical mappings to
|
||||||
|
// coexist instead of overwriting one another on every process switch.
|
||||||
|
CMP(32, MDisp(R11, static_cast<s32>(offsetof(FastEntry, key))), R(EAX));
|
||||||
|
FixupBranch mmu_way0_key_miss = J_CC(CC_NE, Jump::Near);
|
||||||
|
CMP(32, MDisp(R11, static_cast<s32>(offsetof(FastEntry, physical_page))), R(R8));
|
||||||
|
FixupBranch mmu_hit_way0 = J_CC(CC_E, Jump::Near);
|
||||||
|
SetJumpTarget(mmu_way0_key_miss);
|
||||||
|
CMP(32, MDisp(R11, static_cast<s32>(sizeof(FastEntry) + offsetof(FastEntry, key))), R(EAX));
|
||||||
|
FixupBranch mmu_way1_key_miss = J_CC(CC_NE, Jump::Near);
|
||||||
|
CMP(32, MDisp(R11, static_cast<s32>(sizeof(FastEntry) + offsetof(FastEntry, physical_page))),
|
||||||
|
R(R8));
|
||||||
|
FixupBranch mmu_hit_way1 = J_CC(CC_E, Jump::Near);
|
||||||
|
SetJumpTarget(mmu_way1_key_miss);
|
||||||
|
CMP(32, MDisp(R11, static_cast<s32>(2 * sizeof(FastEntry) + offsetof(FastEntry, key))), R(EAX));
|
||||||
|
FixupBranch mmu_way2_key_miss = J_CC(CC_NE, Jump::Near);
|
||||||
|
CMP(32, MDisp(R11,
|
||||||
|
static_cast<s32>(2 * sizeof(FastEntry) + offsetof(FastEntry, physical_page))),
|
||||||
|
R(R8));
|
||||||
|
FixupBranch mmu_hit_way2 = J_CC(CC_E, Jump::Near);
|
||||||
|
SetJumpTarget(mmu_way2_key_miss);
|
||||||
|
CMP(32, MDisp(R11, static_cast<s32>(3 * sizeof(FastEntry) + offsetof(FastEntry, key))), R(EAX));
|
||||||
|
FixupBranch mmu_cache_miss_key = J_CC(CC_NE, Jump::Near);
|
||||||
|
CMP(32, MDisp(R11,
|
||||||
|
static_cast<s32>(3 * sizeof(FastEntry) + offsetof(FastEntry, physical_page))),
|
||||||
|
R(R8));
|
||||||
|
FixupBranch mmu_cache_miss_page = J_CC(CC_NE, Jump::Near);
|
||||||
|
ADD(64, R(R11), Imm8(static_cast<u8>(3 * sizeof(FastEntry))));
|
||||||
|
FixupBranch mmu_way_selected = J(Jump::Near);
|
||||||
|
SetJumpTarget(mmu_hit_way2);
|
||||||
|
ADD(64, R(R11), Imm8(static_cast<u8>(2 * sizeof(FastEntry))));
|
||||||
|
FixupBranch mmu_way2_selected = J(Jump::Near);
|
||||||
|
SetJumpTarget(mmu_hit_way1);
|
||||||
|
ADD(64, R(R11), Imm8(static_cast<u8>(sizeof(FastEntry))));
|
||||||
|
SetJumpTarget(mmu_hit_way0);
|
||||||
|
SetJumpTarget(mmu_way2_selected);
|
||||||
|
SetJumpTarget(mmu_way_selected);
|
||||||
MOV(64, R(R11), MDisp(R11, static_cast<s32>(offsetof(FastEntry, entry))));
|
MOV(64, R(R11), MDisp(R11, static_cast<s32>(offsetof(FastEntry, entry))));
|
||||||
TEST(64, R(R11), R(R11));
|
TEST(64, R(R11), R(R11));
|
||||||
FixupBranch empty_entry = J_CC(CC_Z);
|
FixupBranch empty_entry = J_CC(CC_Z);
|
||||||
JMPptr(R(R11));
|
JMPptr(R(R11));
|
||||||
|
|
||||||
SetJumpTarget(stale_generation);
|
SetJumpTarget(stale_page);
|
||||||
SetJumpTarget(cache_miss);
|
SetJumpTarget(stale_translation_table);
|
||||||
|
SetJumpTarget(unavailable_first_descriptor);
|
||||||
|
SetJumpTarget(changed_first_descriptor);
|
||||||
|
SetJumpTarget(changed_second_descriptor);
|
||||||
|
MOV(32, R(ABI_PARAM4), R(EDX));
|
||||||
|
MOV(32, R(ABI_PARAM3), R(EAX));
|
||||||
|
MOV(32, R(ABI_PARAM2), Imm32(static_cast<u32>(DispatchReason::Translation)));
|
||||||
|
FixupBranch dispatch_reason_translation = J(Jump::Near);
|
||||||
|
SetJumpTarget(no_mmu_cache_miss_key);
|
||||||
|
SetJumpTarget(mmu_cache_miss_key);
|
||||||
|
MOV(32, R(ABI_PARAM4), R(EDX));
|
||||||
|
MOV(32, R(ABI_PARAM3), R(EAX));
|
||||||
|
MOV(32, R(ABI_PARAM2), Imm32(static_cast<u32>(DispatchReason::KeyMiss)));
|
||||||
|
FixupBranch dispatch_reason_key = J(Jump::Near);
|
||||||
|
SetJumpTarget(no_mmu_cache_miss_page);
|
||||||
|
SetJumpTarget(mmu_cache_miss_page);
|
||||||
|
MOV(32, R(ABI_PARAM4), R(EDX));
|
||||||
|
MOV(32, R(ABI_PARAM3), R(EAX));
|
||||||
|
MOV(32, R(ABI_PARAM2), Imm32(static_cast<u32>(DispatchReason::PageMiss)));
|
||||||
|
FixupBranch dispatch_reason_page = J(Jump::Near);
|
||||||
SetJumpTarget(empty_entry);
|
SetJumpTarget(empty_entry);
|
||||||
|
SetJumpTarget(empty_entry_no_mmu);
|
||||||
|
MOV(32, R(ABI_PARAM4), R(EDX));
|
||||||
|
MOV(32, R(ABI_PARAM3), R(EAX));
|
||||||
|
MOV(32, R(ABI_PARAM2), Imm32(static_cast<u32>(DispatchReason::EmptyEntry)));
|
||||||
|
SetJumpTarget(dispatch_reason_translation);
|
||||||
|
SetJumpTarget(dispatch_reason_key);
|
||||||
|
SetJumpTarget(dispatch_reason_page);
|
||||||
MOV(64, R(ABI_PARAM1), ImmPtr(this));
|
MOV(64, R(ABI_PARAM1), ImmPtr(this));
|
||||||
ABI_CallFunction(Dispatch);
|
ABI_CallFunction(Dispatch);
|
||||||
TEST(64, R(RAX), R(RAX));
|
TEST(64, R(RAX), R(RAX));
|
||||||
@@ -245,48 +420,154 @@ u32 ARMJitX64::MakeFastEntryKey(u32 address, u32 control, u32 process_id)
|
|||||||
return address;
|
return address;
|
||||||
}
|
}
|
||||||
|
|
||||||
size_t ARMJitX64::GetFastEntryIndex(u32 key)
|
size_t ARMJitX64::GetFastEntrySetIndex(u32 key)
|
||||||
{
|
{
|
||||||
return ((key >> 1) ^ (key >> 17)) & (FAST_ENTRY_COUNT - 1);
|
return static_cast<u32>(key * 0x9e3779b1U) >> 16;
|
||||||
}
|
}
|
||||||
|
|
||||||
const u8* ARMJitX64::Dispatch(ARMJitX64* jit)
|
const u8* ARMJitX64::Dispatch(ARMJitX64* jit, DispatchReason reason, u32 generated_key,
|
||||||
|
u32 generated_set_offset)
|
||||||
{
|
{
|
||||||
const bool thumb = (jit->m_core.m_cpsr & ARMCore::CPSR_T) != 0;
|
const bool thumb = (jit->m_core.m_cpsr & ARMCore::CPSR_T) != 0;
|
||||||
const u32 key = jit->m_core.m_registers[15] | (thumb ? 1U : 0U);
|
const u32 key = jit->m_core.m_registers[15] | (thumb ? 1U : 0U);
|
||||||
Block* const block = jit->GetOrCompileBlock(key);
|
|
||||||
if (!block || !block->runnable)
|
|
||||||
return nullptr;
|
|
||||||
|
|
||||||
const u32 fast_key =
|
const u32 fast_key =
|
||||||
MakeFastEntryKey(key, jit->m_core.m_cp15.control, jit->m_core.m_cp15.process_id);
|
MakeFastEntryKey(key, jit->m_core.m_cp15.control, jit->m_core.m_cp15.process_id);
|
||||||
FastEntry& fast_entry = jit->m_fast_entries[GetFastEntryIndex(fast_key)];
|
const u32 virtual_page = fast_key >> 10;
|
||||||
fast_entry.key = fast_key;
|
FastTranslationEntry& translation =
|
||||||
fast_entry.tlb_generation = jit->m_core.m_tlb_generation;
|
jit->m_fast_translations[(virtual_page ^ (virtual_page >> 12)) &
|
||||||
fast_entry.entry = block->entry;
|
(FAST_TRANSLATION_ENTRY_COUNT - 1)];
|
||||||
|
const bool mmu_enabled = (jit->m_core.m_cp15.control & 1) != 0;
|
||||||
|
const bool translation_is_current =
|
||||||
|
!mmu_enabled || (translation.tlb_generation == jit->m_core.m_tlb_generation &&
|
||||||
|
translation.virtual_page == virtual_page);
|
||||||
|
const size_t set_index = GetFastEntrySetIndex(fast_key);
|
||||||
|
const size_t set_base = set_index * FAST_ENTRY_WAYS;
|
||||||
|
if (generated_key != fast_key)
|
||||||
|
++jit->m_dispatch_generated_key_mismatch_count;
|
||||||
|
if (generated_set_offset != set_index * FAST_ENTRY_WAYS * sizeof(FastEntry))
|
||||||
|
++jit->m_dispatch_generated_set_mismatch_count;
|
||||||
|
++jit->m_dispatch_slow_count;
|
||||||
|
if (reason == DispatchReason::Translation)
|
||||||
|
++jit->m_dispatch_stale_generation_count;
|
||||||
|
else if (reason == DispatchReason::KeyMiss)
|
||||||
|
++jit->m_dispatch_key_miss_count;
|
||||||
|
else if (reason == DispatchReason::PageMiss)
|
||||||
|
++jit->m_dispatch_page_miss_count;
|
||||||
|
else if (reason == DispatchReason::EmptyEntry)
|
||||||
|
++jit->m_dispatch_empty_entry_count;
|
||||||
|
|
||||||
|
u32 physical_address = 0;
|
||||||
|
const u8* first_descriptor = nullptr;
|
||||||
|
const u8* second_descriptor = nullptr;
|
||||||
|
u32 first_descriptor_value = 0;
|
||||||
|
u32 second_descriptor_value = 0;
|
||||||
|
Block* const block =
|
||||||
|
jit->GetOrCompileBlock(key, &physical_address, &first_descriptor, &first_descriptor_value,
|
||||||
|
&second_descriptor, &second_descriptor_value);
|
||||||
|
if (!block || !block->runnable)
|
||||||
|
return nullptr;
|
||||||
|
const u32 physical_page = physical_address & ~0x3ffU;
|
||||||
|
|
||||||
|
FastEntry* matching_entry = nullptr;
|
||||||
|
FastEntry* empty_entry = nullptr;
|
||||||
|
bool has_other_physical_mapping = false;
|
||||||
|
bool has_same_physical_page = false;
|
||||||
|
for (size_t way = 0; way < FAST_ENTRY_WAYS; ++way)
|
||||||
|
{
|
||||||
|
FastEntry& entry = jit->m_fast_entries[set_base + way];
|
||||||
|
if (entry.key == fast_key && entry.physical_page == physical_page)
|
||||||
|
{
|
||||||
|
matching_entry = &entry;
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
if (entry.key == fast_key && entry.entry != nullptr)
|
||||||
|
has_other_physical_mapping = true;
|
||||||
|
if (entry.physical_page == physical_page && entry.entry != nullptr)
|
||||||
|
has_same_physical_page = true;
|
||||||
|
if (entry.entry == nullptr && empty_entry == nullptr)
|
||||||
|
empty_entry = &entry;
|
||||||
|
}
|
||||||
|
if (translation_is_current && matching_entry == nullptr)
|
||||||
|
{
|
||||||
|
if (has_other_physical_mapping)
|
||||||
|
++jit->m_dispatch_physical_alias_count;
|
||||||
|
else if (empty_entry == nullptr)
|
||||||
|
++jit->m_dispatch_collision_count;
|
||||||
|
}
|
||||||
|
if (reason == DispatchReason::KeyMiss)
|
||||||
|
{
|
||||||
|
if (matching_entry != nullptr)
|
||||||
|
++jit->m_dispatch_key_miss_with_exact_match_count;
|
||||||
|
if (has_other_physical_mapping)
|
||||||
|
++jit->m_dispatch_key_miss_with_physical_alias_count;
|
||||||
|
if (has_same_physical_page)
|
||||||
|
++jit->m_dispatch_key_miss_with_same_physical_page_count;
|
||||||
|
if (empty_entry != nullptr)
|
||||||
|
++jit->m_dispatch_key_miss_with_empty_slot_count;
|
||||||
|
else
|
||||||
|
++jit->m_dispatch_key_miss_with_full_set_count;
|
||||||
|
if (!translation_is_current)
|
||||||
|
++jit->m_dispatch_key_miss_with_stale_translation_count;
|
||||||
|
}
|
||||||
|
|
||||||
|
translation.virtual_page = virtual_page;
|
||||||
|
translation.physical_page = physical_page;
|
||||||
|
translation.tlb_generation = jit->m_core.m_tlb_generation;
|
||||||
|
translation.translation_table_base = jit->m_core.m_cp15.translation_table_base;
|
||||||
|
translation.first_descriptor = first_descriptor;
|
||||||
|
translation.second_descriptor = second_descriptor;
|
||||||
|
translation.first_descriptor_value = first_descriptor_value;
|
||||||
|
translation.second_descriptor_value = second_descriptor_value;
|
||||||
|
FastEntry* fast_entry = matching_entry;
|
||||||
|
if (fast_entry == nullptr)
|
||||||
|
{
|
||||||
|
if (empty_entry != nullptr)
|
||||||
|
{
|
||||||
|
fast_entry = empty_entry;
|
||||||
|
++jit->m_fast_entry_empty_insert_count;
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
u8& next_victim = jit->m_fast_entry_next_victim[set_index];
|
||||||
|
fast_entry = &jit->m_fast_entries[set_base + next_victim];
|
||||||
|
next_victim = static_cast<u8>((next_victim + 1) & (FAST_ENTRY_WAYS - 1));
|
||||||
|
++jit->m_fast_entry_replacement_count;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
fast_entry->key = fast_key;
|
||||||
|
fast_entry->physical_page = physical_page;
|
||||||
|
fast_entry->entry = block->entry;
|
||||||
return block->entry;
|
return block->entry;
|
||||||
}
|
}
|
||||||
|
|
||||||
ARMJitX64::Block* ARMJitX64::GetOrCompileBlock(u32 address)
|
ARMJitX64::Block* ARMJitX64::GetOrCompileBlock(u32 address, u32* physical_address_out,
|
||||||
|
const u8** first_descriptor_out,
|
||||||
|
u32* first_descriptor_value_out,
|
||||||
|
const u8** second_descriptor_out,
|
||||||
|
u32* second_descriptor_value_out)
|
||||||
{
|
{
|
||||||
|
++m_block_lookup_count;
|
||||||
const bool thumb = (address & 1) != 0;
|
const bool thumb = (address & 1) != 0;
|
||||||
const u32 virtual_address = address & ~1U;
|
const u32 virtual_address = address & ~1U;
|
||||||
const u32 physical_address = m_core.TranslateVirtualAddress(virtual_address);
|
const u32 physical_address = m_core.TranslateVirtualAddressForJit(
|
||||||
|
virtual_address, first_descriptor_out, first_descriptor_value_out, second_descriptor_out,
|
||||||
|
second_descriptor_value_out);
|
||||||
|
*physical_address_out = physical_address;
|
||||||
const u64 block_key = (static_cast<u64>(physical_address) << 32) | address;
|
const u64 block_key = (static_cast<u64>(physical_address) << 32) | address;
|
||||||
if (const auto it = m_blocks.find(block_key); it != m_blocks.end())
|
if (const auto it = m_blocks.find(block_key); it != m_blocks.end())
|
||||||
|
{
|
||||||
|
++m_block_cache_hit_count;
|
||||||
return &it->second;
|
return &it->second;
|
||||||
|
}
|
||||||
|
|
||||||
if (IsAlmostFull())
|
if (IsAlmostFull())
|
||||||
{
|
{
|
||||||
if (m_is_running)
|
ClearForCodeSpace();
|
||||||
{
|
|
||||||
m_clear_pending = true;
|
|
||||||
return nullptr;
|
return nullptr;
|
||||||
}
|
}
|
||||||
Clear();
|
|
||||||
}
|
|
||||||
|
|
||||||
Block block = CompileBlock(virtual_address, thumb);
|
Block block = CompileBlock(virtual_address, thumb);
|
||||||
|
++m_lifetime_compiled_block_count;
|
||||||
return &m_blocks.emplace(block_key, block).first->second;
|
return &m_blocks.emplace(block_key, block).first->second;
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -469,11 +750,60 @@ ARMJitX64::Block ARMJitX64::CompileBlock(u32 address, bool thumb)
|
|||||||
return {.entry = entry,
|
return {.entry = entry,
|
||||||
.instruction_count = instruction_count,
|
.instruction_count = instruction_count,
|
||||||
.native_instruction_count = native_instruction_count,
|
.native_instruction_count = native_instruction_count,
|
||||||
.runnable = native_instruction_count != 0};
|
// A fallback-only block is still translated host code: it calls the exact interpreter
|
||||||
|
// helper, accounts the guest instruction, and returns through the native dispatcher.
|
||||||
|
// Rejecting it here made every unsupported hot instruction repeat address translation,
|
||||||
|
// unordered-map lookup and ABI dispatch in C++ before ARMCore interpreted it anyway.
|
||||||
|
// Caching the already-emitted wrapper preserves identical instruction semantics while
|
||||||
|
// removing that redundant lookup path.
|
||||||
|
.runnable = true};
|
||||||
}
|
}
|
||||||
|
|
||||||
bool ARMJitX64::EmitDirectARM(u32 instruction, u32 address, bool* terminal, bool* dispatcher_exit)
|
bool ARMJitX64::EmitDirectARM(u32 instruction, u32 address, bool* terminal, bool* dispatcher_exit)
|
||||||
{
|
{
|
||||||
|
// The IOS scheduler executes a small set of ARM926 system-control writes in virtually every
|
||||||
|
// syscall and interrupt path. Decode those architectural operations once while compiling the
|
||||||
|
// block instead of entering the complete interpreter decoder every time the block runs.
|
||||||
|
//
|
||||||
|
// Keep TLB maintenance terminal: software may have changed a page table immediately before the
|
||||||
|
// MCR, so compiling subsequent guest instructions through the pre-invalidation mapping would be
|
||||||
|
// incorrect. The helper performs the same generation change as ARMCore::WriteCP15 and resumes at
|
||||||
|
// the following ARM instruction through the generated dispatcher.
|
||||||
|
const bool is_mcr_p15 = (instruction & 0x0f100f10) == 0x0e000f10;
|
||||||
|
if ((instruction >> 28) == 0xe && is_mcr_p15)
|
||||||
|
{
|
||||||
|
const u32 opcode1 = (instruction >> 21) & 7;
|
||||||
|
const u32 crn = (instruction >> 16) & 0xf;
|
||||||
|
const u32 rd = (instruction >> 12) & 0xf;
|
||||||
|
const u32 crm = instruction & 0xf;
|
||||||
|
const u32 opcode2 = (instruction >> 5) & 7;
|
||||||
|
|
||||||
|
if (opcode1 == 0 && crn == 3 && rd != 15)
|
||||||
|
{
|
||||||
|
MOV(32, R(EAX), MRegister(rd));
|
||||||
|
MOV(32, MDomainAccessControl(), R(EAX));
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
// ARMCore currently models only WFI and instruction-cache-affecting c7 operations. All other
|
||||||
|
// c7 writes are architecturally harmless in our single-host-thread memory model. In
|
||||||
|
// particular IOS's c7,c6,1 and c7,c10,1 forms are among its hottest privileged instructions.
|
||||||
|
const bool is_wfi = crn == 7 && crm == 0 && opcode2 == 4;
|
||||||
|
const bool affects_instruction_cache = crn == 7 && (crm == 5 || crm == 7);
|
||||||
|
if (opcode1 == 0 && crn == 7 && !is_wfi && !affects_instruction_cache)
|
||||||
|
return true;
|
||||||
|
|
||||||
|
if (opcode1 == 0 && crn == 8)
|
||||||
|
{
|
||||||
|
FlushRegisterCache();
|
||||||
|
MOV(64, R(ABI_PARAM1), ImmPtr(this));
|
||||||
|
MOV(32, R(ABI_PARAM2), Imm32(address));
|
||||||
|
ABI_CallFunction(InvalidateTLB);
|
||||||
|
*terminal = true;
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// MCR p15, 0, Rd, c7, c10, 4 is ARM926 Drain Write Buffer. It is an ordering barrier, not an
|
// MCR p15, 0, Rd, c7, c10, 4 is ARM926 Drain Write Buffer. It is an ordering barrier, not an
|
||||||
// instruction-cache invalidation, and has no additional observable work in this single-host-
|
// instruction-cache invalidation, and has no additional observable work in this single-host-
|
||||||
// thread memory model. IOS executes it in hot synchronization paths.
|
// thread memory model. IOS executes it in hot synchronization paths.
|
||||||
@@ -640,6 +970,31 @@ bool ARMJitX64::EmitDirectARM(u32 instruction, u32 address, bool* terminal, bool
|
|||||||
return true;
|
return true;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// IOS's crypto/bootstrap code contains tight UMULL loops. A single such instruction accounted
|
||||||
|
// for more than a million interpreter fallbacks during the first seconds of a real NAND boot.
|
||||||
|
// Translate the complete ARMv5 long-multiply family here; this keeps the arithmetic and optional
|
||||||
|
// NZ update architectural while eliminating the generic decoder/dispatcher round trip.
|
||||||
|
if ((instruction & 0x0f8000f0) == 0x00800090)
|
||||||
|
{
|
||||||
|
const u32 rd_hi = (instruction >> 16) & 0xf;
|
||||||
|
const u32 rd_lo = (instruction >> 12) & 0xf;
|
||||||
|
const u32 rs = (instruction >> 8) & 0xf;
|
||||||
|
const u32 rm = instruction & 0xf;
|
||||||
|
if (condition == 0xf || rd_hi == 15 || rd_lo == 15 || rs == 15 || rm == 15)
|
||||||
|
return false;
|
||||||
|
|
||||||
|
if (condition == 0xe)
|
||||||
|
return EmitARMMultiplyLong(instruction);
|
||||||
|
|
||||||
|
EmitConditionResult(condition);
|
||||||
|
TEST(32, R(EAX), R(EAX));
|
||||||
|
const FixupBranch predicate_failed = J_CC(CC_Z, Jump::Near);
|
||||||
|
const bool emitted = EmitARMMultiplyLong(instruction);
|
||||||
|
ASSERT(emitted);
|
||||||
|
SetJumpTarget(predicate_failed);
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
if ((instruction & 0x0e000000) == 0x0a000000 && (instruction >> 28) != 0xf)
|
if ((instruction & 0x0e000000) == 0x0a000000 && (instruction >> 28) != 0xf)
|
||||||
{
|
{
|
||||||
const bool link = (instruction & (1U << 24)) != 0;
|
const bool link = (instruction & (1U << 24)) != 0;
|
||||||
@@ -686,7 +1041,7 @@ bool ARMJitX64::EmitDirectARM(u32 instruction, u32 address, bool* terminal, bool
|
|||||||
EmitConditionResult(condition);
|
EmitConditionResult(condition);
|
||||||
TEST(32, R(EAX), R(EAX));
|
TEST(32, R(EAX), R(EAX));
|
||||||
const FixupBranch predicate_failed = J_CC(CC_Z, Jump::Near);
|
const FixupBranch predicate_failed = J_CC(CC_Z, Jump::Near);
|
||||||
const bool emitted = EmitARMMemory(instruction, address);
|
const bool emitted = EmitARMMemory(instruction, address, terminal);
|
||||||
ASSERT(emitted);
|
ASSERT(emitted);
|
||||||
SetJumpTarget(predicate_failed);
|
SetJumpTarget(predicate_failed);
|
||||||
return true;
|
return true;
|
||||||
@@ -707,7 +1062,7 @@ bool ARMJitX64::EmitDirectARM(u32 instruction, u32 address, bool* terminal, bool
|
|||||||
}
|
}
|
||||||
|
|
||||||
if ((instruction & 0x0c000000) == 0x04000000)
|
if ((instruction & 0x0c000000) == 0x04000000)
|
||||||
return EmitARMMemory(instruction, address);
|
return EmitARMMemory(instruction, address, terminal);
|
||||||
|
|
||||||
if ((instruction & 0x0e000090) == 0x00000090)
|
if ((instruction & 0x0e000090) == 0x00000090)
|
||||||
return EmitARMHalfwordMemory(instruction, address);
|
return EmitARMHalfwordMemory(instruction, address);
|
||||||
@@ -722,6 +1077,73 @@ bool ARMJitX64::EmitDirectARM(u32 instruction, u32 address, bool* terminal, bool
|
|||||||
return true;
|
return true;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
bool ARMJitX64::EmitARMMultiplyLong(u32 instruction)
|
||||||
|
{
|
||||||
|
if ((instruction & 0x0f8000f0) != 0x00800090)
|
||||||
|
return false;
|
||||||
|
|
||||||
|
const bool signed_multiply = (instruction & (1U << 22)) != 0;
|
||||||
|
const bool accumulate = (instruction & (1U << 21)) != 0;
|
||||||
|
const bool set_flags = (instruction & (1U << 20)) != 0;
|
||||||
|
const u32 rd_hi = (instruction >> 16) & 0xf;
|
||||||
|
const u32 rd_lo = (instruction >> 12) & 0xf;
|
||||||
|
const u32 rs = (instruction >> 8) & 0xf;
|
||||||
|
const u32 rm = instruction & 0xf;
|
||||||
|
if (rd_hi == 15 || rd_lo == 15 || rs == 15 || rm == 15)
|
||||||
|
return false;
|
||||||
|
|
||||||
|
// Read every guest operand before writing either destination so architecturally tolerated
|
||||||
|
// source/destination aliases behave exactly like ARMCore::ExecuteMultiplyLong.
|
||||||
|
FlushRegisterCache();
|
||||||
|
if (signed_multiply)
|
||||||
|
{
|
||||||
|
MOVSX(64, 32, RAX, MStoredRegister(rm));
|
||||||
|
MOVSX(64, 32, RCX, MStoredRegister(rs));
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
MOV(32, R(EAX), MStoredRegister(rm));
|
||||||
|
MOV(32, R(ECX), MStoredRegister(rs));
|
||||||
|
}
|
||||||
|
IMUL(64, RAX, R(RCX));
|
||||||
|
|
||||||
|
if (accumulate)
|
||||||
|
{
|
||||||
|
MOV(32, R(R8), MStoredRegister(rd_hi));
|
||||||
|
SHL(64, R(R8), Imm8(32));
|
||||||
|
MOV(32, R(R9), MStoredRegister(rd_lo));
|
||||||
|
OR(64, R(R8), R(R9));
|
||||||
|
ADD(64, R(RAX), R(R8));
|
||||||
|
}
|
||||||
|
|
||||||
|
if (set_flags)
|
||||||
|
{
|
||||||
|
TEST(64, R(RAX), R(RAX));
|
||||||
|
SETcc(CC_S, R(R8));
|
||||||
|
SETcc(CC_Z, R(R9));
|
||||||
|
MOVZX(32, 8, R8, R(R8));
|
||||||
|
MOVZX(32, 8, R9, R(R9));
|
||||||
|
SHL(32, R(R8), Imm8(31));
|
||||||
|
SHL(32, R(R9), Imm8(30));
|
||||||
|
}
|
||||||
|
|
||||||
|
MOV(32, MStoredRegister(rd_lo), R(EAX));
|
||||||
|
MOV(64, R(RDX), R(RAX));
|
||||||
|
SHR(64, R(RDX), Imm8(32));
|
||||||
|
MOV(32, MStoredRegister(rd_hi), R(EDX));
|
||||||
|
|
||||||
|
if (set_flags)
|
||||||
|
{
|
||||||
|
MOV(32, R(ECX), MCPSR());
|
||||||
|
AND(32, R(ECX), Imm32(~(ARMCore::CPSR_N | ARMCore::CPSR_Z)));
|
||||||
|
OR(32, R(ECX), R(R8));
|
||||||
|
OR(32, R(ECX), R(R9));
|
||||||
|
MOV(32, MCPSR(), R(ECX));
|
||||||
|
}
|
||||||
|
LoadRegisterCache();
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
bool ARMJitX64::CanEmitARMDataProcessing(u32 instruction) const
|
bool ARMJitX64::CanEmitARMDataProcessing(u32 instruction) const
|
||||||
{
|
{
|
||||||
// The special encodings in the data-processing space stay on the exact interpreter path. The
|
// The special encodings in the data-processing space stay on the exact interpreter path. The
|
||||||
@@ -759,24 +1181,18 @@ bool ARMJitX64::CanEmitARMDataProcessing(u32 instruction) const
|
|||||||
const bool shifted_register_operand = !immediate && (instruction & 0xff0) != 0;
|
const bool shifted_register_operand = !immediate && (instruction & 0xff0) != 0;
|
||||||
const bool shift_by_register = shifted_register_operand && (instruction & (1U << 4)) != 0;
|
const bool shift_by_register = shifted_register_operand && (instruction & (1U << 4)) != 0;
|
||||||
const u32 rs = (instruction >> 8) & 0xf;
|
const u32 rs = (instruction >> 8) & 0xf;
|
||||||
// Carry-out from a shifted operand is only architecturally visible for logical flag-setting
|
// Register-controlled shifts have several ARM-only carry corner cases for counts >= 32. Keep
|
||||||
// operations. Keep those on the interpreter for now; all non-flag-setting ALU forms can use the
|
// only those flag-setting forms on the interpreter; immediate shifts can materialize their exact
|
||||||
// value-only native shifter exactly.
|
// shifter carry cheaply below.
|
||||||
if (shifted_register_operand && (set_flags || !writes_result))
|
if (shift_by_register && (set_flags || !writes_result))
|
||||||
return false;
|
return false;
|
||||||
if (shift_by_register && rs == 15)
|
if (shift_by_register && rs == 15)
|
||||||
return false;
|
return false;
|
||||||
|
|
||||||
const bool logical = opcode == 0 || opcode == 1 || opcode == 8 || opcode == 9 || opcode == 0xc ||
|
|
||||||
opcode == 0xd || opcode == 0xe || opcode == 0xf;
|
|
||||||
const u32 rotate = ((instruction >> 8) & 0xf) * 2;
|
|
||||||
if (logical && (set_flags || !writes_result) && immediate && rotate != 0)
|
|
||||||
return false;
|
|
||||||
|
|
||||||
return true;
|
return true;
|
||||||
}
|
}
|
||||||
|
|
||||||
bool ARMJitX64::EmitARMMemory(u32 instruction, u32 address)
|
bool ARMJitX64::EmitARMMemory(u32 instruction, u32 address, bool* terminal)
|
||||||
{
|
{
|
||||||
if (!CanEmitARMMemory(instruction))
|
if (!CanEmitARMMemory(instruction))
|
||||||
return false;
|
return false;
|
||||||
@@ -878,8 +1294,22 @@ bool ARMJitX64::EmitARMMemory(u32 instruction, u32 address)
|
|||||||
SHL(32, R(ECX), Imm8(3));
|
SHL(32, R(ECX), Imm8(3));
|
||||||
ROR(32, R(EAX), R(ECX));
|
ROR(32, R(EAX), R(ECX));
|
||||||
}
|
}
|
||||||
|
if (rd == 15)
|
||||||
|
{
|
||||||
|
// ARMv5 LDR PC is an interworking branch. Preserve bit zero as the new Thumb state and
|
||||||
|
// branch to the aligned target. This is the exact operation used by Starlet's high IRQ and
|
||||||
|
// reset vectors, and keeping it in native code avoids a generic decoder call per interrupt.
|
||||||
|
MOV(32, R(ABI_PARAM3), R(EAX));
|
||||||
|
MOV(32, R(ABI_PARAM2), Imm32(address));
|
||||||
|
MOV(64, R(ABI_PARAM1), ImmPtr(this));
|
||||||
|
ABI_CallFunction(LoadPC);
|
||||||
|
*terminal = true;
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
MOV(32, MRegister(rd), R(EAX));
|
MOV(32, MRegister(rd), R(EAX));
|
||||||
}
|
}
|
||||||
|
}
|
||||||
else
|
else
|
||||||
{
|
{
|
||||||
MOV(32, R(EDX), MRegister(rd));
|
MOV(32, R(EDX), MRegister(rd));
|
||||||
@@ -907,7 +1337,10 @@ bool ARMJitX64::CanEmitARMMemory(u32 instruction) const
|
|||||||
const u32 rd = (instruction >> 12) & 0xf;
|
const u32 rd = (instruction >> 12) & 0xf;
|
||||||
const bool register_offset = (instruction & (1U << 25)) != 0;
|
const bool register_offset = (instruction & (1U << 25)) != 0;
|
||||||
const u32 rm = instruction & 0xf;
|
const u32 rm = instruction & 0xf;
|
||||||
return rd != 15 && !(rn == 15 && (!preindex || writeback)) && !(load && writeback && rn == rd) &&
|
const bool direct_pc_load = load && rd == 15 && (instruction >> 28) == 0xe && preindex &&
|
||||||
|
!writeback && (instruction & (1U << 22)) == 0;
|
||||||
|
return (rd != 15 || direct_pc_load) && !(rn == 15 && (!preindex || writeback)) &&
|
||||||
|
!(load && writeback && rn == rd) &&
|
||||||
!(register_offset && (instruction & (1U << 4)) != 0) && !(register_offset && rm == 15);
|
!(register_offset && (instruction & (1U << 4)) != 0) && !(register_offset && rm == 15);
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -1123,11 +1556,21 @@ void ARMJitX64::EmitARMDataProcessing(u32 instruction)
|
|||||||
const u32 rd = (instruction >> 12) & 0xf;
|
const u32 rd = (instruction >> 12) & 0xf;
|
||||||
const u32 rm = instruction & 0xf;
|
const u32 rm = instruction & 0xf;
|
||||||
const bool writes_result = opcode < 8 || opcode > 0xb;
|
const bool writes_result = opcode < 8 || opcode > 0xb;
|
||||||
|
const bool logical = opcode == 0 || opcode == 1 || opcode == 8 || opcode == 9 || opcode == 0xc ||
|
||||||
|
opcode == 0xd || opcode == 0xe || opcode == 0xf;
|
||||||
|
const bool logical_flags = logical && (set_flags || !writes_result);
|
||||||
|
bool logical_carry_known = false;
|
||||||
|
|
||||||
if (immediate)
|
if (immediate)
|
||||||
{
|
{
|
||||||
const u32 rotate = ((instruction >> 8) & 0xf) * 2;
|
const u32 rotate = ((instruction >> 8) & 0xf) * 2;
|
||||||
MOV(32, R(EDX), Imm32(std::rotr(instruction & 0xff, rotate)));
|
const u32 operand = std::rotr(instruction & 0xff, rotate);
|
||||||
|
MOV(32, R(EDX), Imm32(operand));
|
||||||
|
if (logical_flags && rotate != 0)
|
||||||
|
{
|
||||||
|
MOV(32, R(R10), Imm32(operand >> 31));
|
||||||
|
logical_carry_known = true;
|
||||||
|
}
|
||||||
}
|
}
|
||||||
else
|
else
|
||||||
{
|
{
|
||||||
@@ -1168,6 +1611,21 @@ void ARMJitX64::EmitARMDataProcessing(u32 instruction)
|
|||||||
else
|
else
|
||||||
{
|
{
|
||||||
const u32 amount = (instruction >> 7) & 0x1f;
|
const u32 amount = (instruction >> 7) & 0x1f;
|
||||||
|
if (logical_flags)
|
||||||
|
{
|
||||||
|
// ARM's logical S forms take C from the barrel shifter. Save that source bit before EDX
|
||||||
|
// is shifted; rotate zero preserves the old C only for the unshifted LSL #0 encoding,
|
||||||
|
// which never enters this branch.
|
||||||
|
MOV(32, R(R10), R(EDX));
|
||||||
|
if (shift_type == 0)
|
||||||
|
SHR(32, R(R10), Imm8(32 - amount));
|
||||||
|
else if (shift_type == 1 || shift_type == 2)
|
||||||
|
SHR(32, R(R10), Imm8(amount == 0 ? 31 : amount - 1));
|
||||||
|
else if (amount != 0)
|
||||||
|
SHR(32, R(R10), Imm8(amount - 1));
|
||||||
|
AND(32, R(R10), Imm8(1));
|
||||||
|
logical_carry_known = true;
|
||||||
|
}
|
||||||
if (shift_type == 0)
|
if (shift_type == 0)
|
||||||
{
|
{
|
||||||
if (amount != 0)
|
if (amount != 0)
|
||||||
@@ -1253,6 +1711,8 @@ void ARMJitX64::EmitARMDataProcessing(u32 instruction)
|
|||||||
{
|
{
|
||||||
if (arithmetic)
|
if (arithmetic)
|
||||||
EmitArithmeticFlags(opcode == 2 || opcode == 3 || opcode == 0xa);
|
EmitArithmeticFlags(opcode == 2 || opcode == 3 || opcode == 0xa);
|
||||||
|
else if (logical_carry_known)
|
||||||
|
EmitLogicalFlagsWithCarry(EAX, R10);
|
||||||
else
|
else
|
||||||
EmitLogicalFlags(EAX);
|
EmitLogicalFlags(EAX);
|
||||||
}
|
}
|
||||||
@@ -1869,6 +2329,8 @@ void ARMJitX64::EmitFastmemAddress(std::vector<FixupBranch>* slow_paths, u32 acc
|
|||||||
{
|
{
|
||||||
CMP(32, R(EDX), Imm32(0x00));
|
CMP(32, R(EDX), Imm32(0x00));
|
||||||
const FixupBranch profiled_read_page_00 = J_CC(CC_E, Jump::Near);
|
const FixupBranch profiled_read_page_00 = J_CC(CC_E, Jump::Near);
|
||||||
|
CMP(32, R(EDX), Imm32(0x10));
|
||||||
|
const FixupBranch irq_vector_read_page_10 = J_CC(CC_E, Jump::Near);
|
||||||
CMP(32, R(EDX), Imm32(0x12));
|
CMP(32, R(EDX), Imm32(0x12));
|
||||||
const FixupBranch profiled_read_page_12 = J_CC(CC_E, Jump::Near);
|
const FixupBranch profiled_read_page_12 = J_CC(CC_E, Jump::Near);
|
||||||
CMP(32, R(EDX), Imm32(0x14));
|
CMP(32, R(EDX), Imm32(0x14));
|
||||||
@@ -1878,6 +2340,7 @@ void ARMJitX64::EmitFastmemAddress(std::vector<FixupBranch>* slow_paths, u32 acc
|
|||||||
CMP(32, R(EDX), Imm32(0x1e));
|
CMP(32, R(EDX), Imm32(0x1e));
|
||||||
slow_paths->push_back(J_CC(CC_NE, Jump::Near));
|
slow_paths->push_back(J_CC(CC_NE, Jump::Near));
|
||||||
SetJumpTarget(profiled_read_page_00);
|
SetJumpTarget(profiled_read_page_00);
|
||||||
|
SetJumpTarget(irq_vector_read_page_10);
|
||||||
SetJumpTarget(profiled_read_page_12);
|
SetJumpTarget(profiled_read_page_12);
|
||||||
SetJumpTarget(profiled_read_page_14);
|
SetJumpTarget(profiled_read_page_14);
|
||||||
SetJumpTarget(profiled_read_page_19);
|
SetJumpTarget(profiled_read_page_19);
|
||||||
@@ -2368,6 +2831,11 @@ OpArg ARMJitX64::MExecutedInstructions() const
|
|||||||
return MDisp(R15, m_executed_instructions_offset);
|
return MDisp(R15, m_executed_instructions_offset);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
OpArg ARMJitX64::MDomainAccessControl() const
|
||||||
|
{
|
||||||
|
return MDisp(R15, m_domain_access_control_offset);
|
||||||
|
}
|
||||||
|
|
||||||
void ARMJitX64::FallbackThumb(ARMJitX64* jit, u16 instruction, u32 address)
|
void ARMJitX64::FallbackThumb(ARMJitX64* jit, u16 instruction, u32 address)
|
||||||
{
|
{
|
||||||
ARMCore* const core = &jit->m_core;
|
ARMCore* const core = &jit->m_core;
|
||||||
@@ -2417,6 +2885,23 @@ void ARMJitX64::WritePSR(ARMJitX64* jit, u32 spsr, u32 field_mask, u32 value)
|
|||||||
jit->m_core.WritePSR(spsr != 0, field_mask, value);
|
jit->m_core.WritePSR(spsr != 0, field_mask, value);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
void ARMJitX64::InvalidateTLB(ARMJitX64* jit, u32 address)
|
||||||
|
{
|
||||||
|
ARMCore& core = jit->m_core;
|
||||||
|
core.m_instruction_address = address;
|
||||||
|
core.m_pc_written = false;
|
||||||
|
core.InvalidateTLB();
|
||||||
|
core.m_registers[15] = address + 4;
|
||||||
|
}
|
||||||
|
|
||||||
|
void ARMJitX64::LoadPC(ARMJitX64* jit, u32 address, u32 target)
|
||||||
|
{
|
||||||
|
ARMCore& core = jit->m_core;
|
||||||
|
core.m_instruction_address = address;
|
||||||
|
core.m_pc_written = false;
|
||||||
|
core.WritePC(target, true);
|
||||||
|
}
|
||||||
|
|
||||||
void ARMJitX64::ExecuteUserBankBlockTransfer(ARMJitX64* jit, u32 instruction, u32 address)
|
void ARMJitX64::ExecuteUserBankBlockTransfer(ARMJitX64* jit, u32 instruction, u32 address)
|
||||||
{
|
{
|
||||||
ARMCore& core = jit->m_core;
|
ARMCore& core = jit->m_core;
|
||||||
@@ -2444,6 +2929,8 @@ u32 ARMJitX64::ReadMemorySlow(ARMJitX64* jit, u32 physical_address, u32 access_s
|
|||||||
u32 byte_offset)
|
u32 byte_offset)
|
||||||
{
|
{
|
||||||
++jit->m_slow_read_count;
|
++jit->m_slow_read_count;
|
||||||
|
if ((jit->m_slow_read_count & 0xff) == 0)
|
||||||
|
++jit->m_slow_memory_address_samples[physical_address & ~3ULL];
|
||||||
switch (ClassifySlowMemoryAddress(physical_address))
|
switch (ClassifySlowMemoryAddress(physical_address))
|
||||||
{
|
{
|
||||||
case SlowMemoryRegion::RAM:
|
case SlowMemoryRegion::RAM:
|
||||||
@@ -2475,6 +2962,8 @@ u32 ARMJitX64::ReadMemorySlow(ARMJitX64* jit, u32 physical_address, u32 access_s
|
|||||||
void ARMJitX64::WriteMemorySlow(ARMJitX64* jit, u32 physical_address, u32 access_size, u32 value)
|
void ARMJitX64::WriteMemorySlow(ARMJitX64* jit, u32 physical_address, u32 access_size, u32 value)
|
||||||
{
|
{
|
||||||
++jit->m_slow_write_count;
|
++jit->m_slow_write_count;
|
||||||
|
if ((jit->m_slow_write_count & 0xff) == 0)
|
||||||
|
++jit->m_slow_memory_address_samples[(1ULL << 32) | (physical_address & ~3ULL)];
|
||||||
switch (ClassifySlowMemoryAddress(physical_address))
|
switch (ClassifySlowMemoryAddress(physical_address))
|
||||||
{
|
{
|
||||||
case SlowMemoryRegion::RAM:
|
case SlowMemoryRegion::RAM:
|
||||||
@@ -2505,6 +2994,18 @@ void ARMJitX64::WriteMemorySlow(ARMJitX64* jit, u32 physical_address, u32 access
|
|||||||
core.m_bus.Write32(physical_address & ~3U, value);
|
core.m_bus.Write32(physical_address & ~3U, value);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
std::vector<std::pair<u64, u64>> ARMJitX64::TakeHotSlowMemorySamples(size_t maximum_count)
|
||||||
|
{
|
||||||
|
std::vector<std::pair<u64, u64>> sorted(m_slow_memory_address_samples.begin(),
|
||||||
|
m_slow_memory_address_samples.end());
|
||||||
|
m_slow_memory_address_samples.clear();
|
||||||
|
std::ranges::sort(sorted, {}, [](const auto& entry) { return entry.second; });
|
||||||
|
if (sorted.size() > maximum_count)
|
||||||
|
sorted.erase(sorted.begin(), sorted.end() - maximum_count);
|
||||||
|
std::ranges::reverse(sorted);
|
||||||
|
return sorted;
|
||||||
|
}
|
||||||
|
|
||||||
u32 ARMJitX64::TranslateAddress(ARMJitX64* jit, u32 address)
|
u32 ARMJitX64::TranslateAddress(ARMJitX64* jit, u32 address)
|
||||||
{
|
{
|
||||||
++jit->m_address_translation_count;
|
++jit->m_address_translation_count;
|
||||||
|
|||||||
@@ -35,6 +35,53 @@ public:
|
|||||||
u64 GetExecutedInstructions() const { return m_executed_instructions; }
|
u64 GetExecutedInstructions() const { return m_executed_instructions; }
|
||||||
u64 GetNativeExecutedInstructions() const { return m_native_executed_instructions; }
|
u64 GetNativeExecutedInstructions() const { return m_native_executed_instructions; }
|
||||||
size_t GetCompiledBlockCount() const { return m_blocks.size(); }
|
size_t GetCompiledBlockCount() const { return m_blocks.size(); }
|
||||||
|
u64 GetLifetimeCompiledBlockCount() const { return m_lifetime_compiled_block_count; }
|
||||||
|
u64 GetBlockLookupCount() const { return m_block_lookup_count; }
|
||||||
|
u64 GetBlockCacheHitCount() const { return m_block_cache_hit_count; }
|
||||||
|
u64 GetDispatchSlowCount() const { return m_dispatch_slow_count; }
|
||||||
|
u64 GetDispatchStaleGenerationCount() const { return m_dispatch_stale_generation_count; }
|
||||||
|
u64 GetDispatchPhysicalAliasCount() const { return m_dispatch_physical_alias_count; }
|
||||||
|
u64 GetDispatchCollisionCount() const { return m_dispatch_collision_count; }
|
||||||
|
u64 GetDispatchKeyMissCount() const { return m_dispatch_key_miss_count; }
|
||||||
|
u64 GetDispatchPageMissCount() const { return m_dispatch_page_miss_count; }
|
||||||
|
u64 GetDispatchEmptyEntryCount() const { return m_dispatch_empty_entry_count; }
|
||||||
|
u64 GetDispatchGeneratedKeyMismatchCount() const
|
||||||
|
{
|
||||||
|
return m_dispatch_generated_key_mismatch_count;
|
||||||
|
}
|
||||||
|
u64 GetDispatchGeneratedSetMismatchCount() const
|
||||||
|
{
|
||||||
|
return m_dispatch_generated_set_mismatch_count;
|
||||||
|
}
|
||||||
|
u64 GetDispatchKeyMissWithExactMatchCount() const
|
||||||
|
{
|
||||||
|
return m_dispatch_key_miss_with_exact_match_count;
|
||||||
|
}
|
||||||
|
u64 GetDispatchKeyMissWithPhysicalAliasCount() const
|
||||||
|
{
|
||||||
|
return m_dispatch_key_miss_with_physical_alias_count;
|
||||||
|
}
|
||||||
|
u64 GetDispatchKeyMissWithSamePhysicalPageCount() const
|
||||||
|
{
|
||||||
|
return m_dispatch_key_miss_with_same_physical_page_count;
|
||||||
|
}
|
||||||
|
u64 GetDispatchKeyMissWithEmptySlotCount() const
|
||||||
|
{
|
||||||
|
return m_dispatch_key_miss_with_empty_slot_count;
|
||||||
|
}
|
||||||
|
u64 GetDispatchKeyMissWithFullSetCount() const
|
||||||
|
{
|
||||||
|
return m_dispatch_key_miss_with_full_set_count;
|
||||||
|
}
|
||||||
|
u64 GetDispatchKeyMissWithStaleTranslationCount() const
|
||||||
|
{
|
||||||
|
return m_dispatch_key_miss_with_stale_translation_count;
|
||||||
|
}
|
||||||
|
u64 GetFastEntryEmptyInsertCount() const { return m_fast_entry_empty_insert_count; }
|
||||||
|
u64 GetFastEntryReplacementCount() const { return m_fast_entry_replacement_count; }
|
||||||
|
u64 GetInstructionCacheClearCount() const { return m_instruction_cache_clear_count; }
|
||||||
|
u64 GetCodeSpaceClearCount() const { return m_code_space_clear_count; }
|
||||||
|
u64 GetDiscardedBlockCount() const { return m_discarded_block_count; }
|
||||||
u64 GetAddressTranslationCount() const { return m_address_translation_count; }
|
u64 GetAddressTranslationCount() const { return m_address_translation_count; }
|
||||||
u64 GetSlowReadCount() const { return m_slow_read_count; }
|
u64 GetSlowReadCount() const { return m_slow_read_count; }
|
||||||
u64 GetSlowWriteCount() const { return m_slow_write_count; }
|
u64 GetSlowWriteCount() const { return m_slow_write_count; }
|
||||||
@@ -62,6 +109,7 @@ public:
|
|||||||
m_slow_sram_page_write_access_count[page] :
|
m_slow_sram_page_write_access_count[page] :
|
||||||
0;
|
0;
|
||||||
}
|
}
|
||||||
|
std::vector<std::pair<u64, u64>> TakeHotSlowMemorySamples(size_t maximum_count);
|
||||||
|
|
||||||
private:
|
private:
|
||||||
using RunEntry = u64 (*)(u32);
|
using RunEntry = u64 (*)(u32);
|
||||||
@@ -84,21 +132,54 @@ private:
|
|||||||
struct FastEntry
|
struct FastEntry
|
||||||
{
|
{
|
||||||
u32 key = 0xffffffff;
|
u32 key = 0xffffffff;
|
||||||
u32 tlb_generation = 0;
|
u32 physical_page = 0xffffffff;
|
||||||
const u8* entry = nullptr;
|
const u8* entry = nullptr;
|
||||||
};
|
};
|
||||||
|
static_assert(sizeof(FastEntry) == 16);
|
||||||
|
|
||||||
|
// TLB maintenance invalidates translations, not every decoded instruction on a page. Keep one
|
||||||
|
// generation-tagged physical identity per 1 KiB ARM926 TLB granule so the first block after a
|
||||||
|
// flush performs the page-table walk and the remaining blocks can safely retain native code.
|
||||||
|
struct alignas(64) FastTranslationEntry
|
||||||
|
{
|
||||||
|
u32 virtual_page = 0xffffffff;
|
||||||
|
u32 physical_page = 0xffffffff;
|
||||||
|
u32 tlb_generation = 0;
|
||||||
|
u32 translation_table_base = 0xffffffff;
|
||||||
|
const u8* first_descriptor = nullptr;
|
||||||
|
const u8* second_descriptor = nullptr;
|
||||||
|
u32 first_descriptor_value = 0;
|
||||||
|
u32 second_descriptor_value = 0;
|
||||||
|
std::array<u32, 6> padding{};
|
||||||
|
};
|
||||||
|
static_assert(sizeof(FastTranslationEntry) == 64);
|
||||||
|
|
||||||
|
enum class DispatchReason : u32
|
||||||
|
{
|
||||||
|
Translation,
|
||||||
|
KeyMiss,
|
||||||
|
PageMiss,
|
||||||
|
EmptyEntry,
|
||||||
|
};
|
||||||
|
|
||||||
void PoisonMemory() override;
|
void PoisonMemory() override;
|
||||||
|
void RequestClear();
|
||||||
|
void ClearCodeCache();
|
||||||
|
void ClearForCodeSpace();
|
||||||
void GenerateDispatcher();
|
void GenerateDispatcher();
|
||||||
static u32 MakeFastEntryKey(u32 address, u32 control, u32 process_id);
|
static u32 MakeFastEntryKey(u32 address, u32 control, u32 process_id);
|
||||||
static size_t GetFastEntryIndex(u32 key);
|
static size_t GetFastEntrySetIndex(u32 key);
|
||||||
static const u8* Dispatch(ARMJitX64* jit);
|
static const u8* Dispatch(ARMJitX64* jit, DispatchReason reason, u32 generated_key,
|
||||||
Block* GetOrCompileBlock(u32 address);
|
u32 generated_set_offset);
|
||||||
|
Block* GetOrCompileBlock(u32 address, u32* physical_address_out,
|
||||||
|
const u8** first_descriptor_out, u32* first_descriptor_value_out,
|
||||||
|
const u8** second_descriptor_out, u32* second_descriptor_value_out);
|
||||||
Block CompileBlock(u32 address, bool thumb);
|
Block CompileBlock(u32 address, bool thumb);
|
||||||
bool EmitDirectARM(u32 instruction, u32 address, bool* terminal, bool* dispatcher_exit);
|
bool EmitDirectARM(u32 instruction, u32 address, bool* terminal, bool* dispatcher_exit);
|
||||||
|
bool EmitARMMultiplyLong(u32 instruction);
|
||||||
bool CanEmitARMDataProcessing(u32 instruction) const;
|
bool CanEmitARMDataProcessing(u32 instruction) const;
|
||||||
bool CanEmitARMMemory(u32 instruction) const;
|
bool CanEmitARMMemory(u32 instruction) const;
|
||||||
bool EmitARMMemory(u32 instruction, u32 address);
|
bool EmitARMMemory(u32 instruction, u32 address, bool* terminal);
|
||||||
bool EmitARMHalfwordMemory(u32 instruction, u32 address);
|
bool EmitARMHalfwordMemory(u32 instruction, u32 address);
|
||||||
bool EmitARMBlockTransfer(u32 instruction, u32 address, bool* terminal);
|
bool EmitARMBlockTransfer(u32 instruction, u32 address, bool* terminal);
|
||||||
void EmitARMDataProcessing(u32 instruction);
|
void EmitARMDataProcessing(u32 instruction);
|
||||||
@@ -137,11 +218,14 @@ private:
|
|||||||
Gen::OpArg MWaitingForMemoryPoll() const;
|
Gen::OpArg MWaitingForMemoryPoll() const;
|
||||||
Gen::OpArg MYieldRequested() const;
|
Gen::OpArg MYieldRequested() const;
|
||||||
Gen::OpArg MExecutedInstructions() const;
|
Gen::OpArg MExecutedInstructions() const;
|
||||||
|
Gen::OpArg MDomainAccessControl() const;
|
||||||
|
|
||||||
static void FallbackThumb(ARMJitX64* jit, u16 instruction, u32 address);
|
static void FallbackThumb(ARMJitX64* jit, u16 instruction, u32 address);
|
||||||
static void FallbackARM(ARMJitX64* jit, u32 instruction, u32 address);
|
static void FallbackARM(ARMJitX64* jit, u32 instruction, u32 address);
|
||||||
static u32 ReadSPSR(ARMJitX64* jit);
|
static u32 ReadSPSR(ARMJitX64* jit);
|
||||||
static void WritePSR(ARMJitX64* jit, u32 spsr, u32 field_mask, u32 value);
|
static void WritePSR(ARMJitX64* jit, u32 spsr, u32 field_mask, u32 value);
|
||||||
|
static void InvalidateTLB(ARMJitX64* jit, u32 address);
|
||||||
|
static void LoadPC(ARMJitX64* jit, u32 address, u32 target);
|
||||||
static void ExecuteUserBankBlockTransfer(ARMJitX64* jit, u32 instruction, u32 address);
|
static void ExecuteUserBankBlockTransfer(ARMJitX64* jit, u32 instruction, u32 address);
|
||||||
static void ExecuteThumbPushPop(ARMJitX64* jit, u16 instruction, u32 address);
|
static void ExecuteThumbPushPop(ARMJitX64* jit, u16 instruction, u32 address);
|
||||||
static void ExceptionReturn(ARMJitX64* jit, u32 target);
|
static void ExceptionReturn(ARMJitX64* jit, u32 target);
|
||||||
@@ -154,11 +238,17 @@ private:
|
|||||||
|
|
||||||
static constexpr size_t CODE_SIZE = 32 * 1024 * 1024;
|
static constexpr size_t CODE_SIZE = 32 * 1024 * 1024;
|
||||||
static constexpr u32 MAX_BLOCK_INSTRUCTIONS = 32;
|
static constexpr u32 MAX_BLOCK_INSTRUCTIONS = 32;
|
||||||
static constexpr size_t FAST_ENTRY_COUNT = 1 << 16;
|
static constexpr size_t FAST_ENTRY_SET_COUNT = 1 << 16;
|
||||||
|
static constexpr size_t FAST_ENTRY_WAYS = 4;
|
||||||
|
static constexpr size_t FAST_ENTRY_COUNT = FAST_ENTRY_SET_COUNT * FAST_ENTRY_WAYS;
|
||||||
|
static_assert((FAST_ENTRY_WAYS & (FAST_ENTRY_WAYS - 1)) == 0);
|
||||||
|
static constexpr size_t FAST_TRANSLATION_ENTRY_COUNT = 1 << 12;
|
||||||
|
|
||||||
ARMCore& m_core;
|
ARMCore& m_core;
|
||||||
std::unordered_map<u64, Block> m_blocks;
|
std::unordered_map<u64, Block> m_blocks;
|
||||||
std::array<FastEntry, FAST_ENTRY_COUNT> m_fast_entries{};
|
alignas(64) std::array<FastEntry, FAST_ENTRY_COUNT> m_fast_entries{};
|
||||||
|
std::array<u8, FAST_ENTRY_SET_COUNT> m_fast_entry_next_victim{};
|
||||||
|
std::array<FastTranslationEntry, FAST_TRANSLATION_ENTRY_COUNT> m_fast_translations{};
|
||||||
u8* m_fastmem_base = nullptr;
|
u8* m_fastmem_base = nullptr;
|
||||||
u8* m_sram_base = nullptr;
|
u8* m_sram_base = nullptr;
|
||||||
const bool* m_boot0_mapped = nullptr;
|
const bool* m_boot0_mapped = nullptr;
|
||||||
@@ -169,6 +259,29 @@ private:
|
|||||||
u8* m_block_code_begin = nullptr;
|
u8* m_block_code_begin = nullptr;
|
||||||
u64 m_executed_instructions = 0;
|
u64 m_executed_instructions = 0;
|
||||||
u64 m_native_executed_instructions = 0;
|
u64 m_native_executed_instructions = 0;
|
||||||
|
u64 m_lifetime_compiled_block_count = 0;
|
||||||
|
u64 m_block_lookup_count = 0;
|
||||||
|
u64 m_block_cache_hit_count = 0;
|
||||||
|
u64 m_dispatch_slow_count = 0;
|
||||||
|
u64 m_dispatch_stale_generation_count = 0;
|
||||||
|
u64 m_dispatch_physical_alias_count = 0;
|
||||||
|
u64 m_dispatch_collision_count = 0;
|
||||||
|
u64 m_dispatch_key_miss_count = 0;
|
||||||
|
u64 m_dispatch_page_miss_count = 0;
|
||||||
|
u64 m_dispatch_empty_entry_count = 0;
|
||||||
|
u64 m_dispatch_generated_key_mismatch_count = 0;
|
||||||
|
u64 m_dispatch_generated_set_mismatch_count = 0;
|
||||||
|
u64 m_dispatch_key_miss_with_exact_match_count = 0;
|
||||||
|
u64 m_dispatch_key_miss_with_physical_alias_count = 0;
|
||||||
|
u64 m_dispatch_key_miss_with_same_physical_page_count = 0;
|
||||||
|
u64 m_dispatch_key_miss_with_empty_slot_count = 0;
|
||||||
|
u64 m_dispatch_key_miss_with_full_set_count = 0;
|
||||||
|
u64 m_dispatch_key_miss_with_stale_translation_count = 0;
|
||||||
|
u64 m_fast_entry_empty_insert_count = 0;
|
||||||
|
u64 m_fast_entry_replacement_count = 0;
|
||||||
|
u64 m_instruction_cache_clear_count = 0;
|
||||||
|
u64 m_code_space_clear_count = 0;
|
||||||
|
u64 m_discarded_block_count = 0;
|
||||||
u64 m_address_translation_count = 0;
|
u64 m_address_translation_count = 0;
|
||||||
u64 m_slow_read_count = 0;
|
u64 m_slow_read_count = 0;
|
||||||
u64 m_slow_write_count = 0;
|
u64 m_slow_write_count = 0;
|
||||||
@@ -184,6 +297,7 @@ private:
|
|||||||
u64 m_slow_sram_high_access_count = 0;
|
u64 m_slow_sram_high_access_count = 0;
|
||||||
std::array<u64, 32> m_slow_sram_page_read_access_count{};
|
std::array<u64, 32> m_slow_sram_page_read_access_count{};
|
||||||
std::array<u64, 32> m_slow_sram_page_write_access_count{};
|
std::array<u64, 32> m_slow_sram_page_write_access_count{};
|
||||||
|
std::unordered_map<u64, u64> m_slow_memory_address_samples;
|
||||||
bool m_is_running = false;
|
bool m_is_running = false;
|
||||||
bool m_clear_pending = false;
|
bool m_clear_pending = false;
|
||||||
s32 m_registers_offset = 0;
|
s32 m_registers_offset = 0;
|
||||||
@@ -195,6 +309,8 @@ private:
|
|||||||
s32 m_yield_requested_offset = 0;
|
s32 m_yield_requested_offset = 0;
|
||||||
s32 m_executed_instructions_offset = 0;
|
s32 m_executed_instructions_offset = 0;
|
||||||
s32 m_control_offset = 0;
|
s32 m_control_offset = 0;
|
||||||
|
s32 m_translation_table_base_offset = 0;
|
||||||
|
s32 m_domain_access_control_offset = 0;
|
||||||
s32 m_process_id_offset = 0;
|
s32 m_process_id_offset = 0;
|
||||||
s32 m_tlb_generation_offset = 0;
|
s32 m_tlb_generation_offset = 0;
|
||||||
u32 m_compile_instruction_count = 0;
|
u32 m_compile_instruction_count = 0;
|
||||||
|
|||||||
@@ -296,6 +296,32 @@ void Starlet::RunSlice(s64 cycles_late)
|
|||||||
m_core->GetCPSR(), m_system.GetWiiIPC().ReadStarletRegister(0x38),
|
m_core->GetCPSR(), m_system.GetWiiIPC().ReadStarletRegister(0x38),
|
||||||
m_system.GetWiiIPC().ReadStarletRegister(0x3c),
|
m_system.GetWiiIPC().ReadStarletRegister(0x3c),
|
||||||
m_system.GetWiiIPC().ReadStarletRegister(0x40));
|
m_system.GetWiiIPC().ReadStarletRegister(0x40));
|
||||||
|
INFO_LOG_FMT(IOS,
|
||||||
|
"Starlet JIT cache: live-blocks={} lifetime-compiles={} lookups={} hits={} "
|
||||||
|
"slow-dispatches={} stale-generations={} physical-aliases={} collisions={} "
|
||||||
|
"key-misses={} page-misses={} empty-entries={} "
|
||||||
|
"generated-key-mismatches={} generated-set-mismatches={} exact-key-misses={} "
|
||||||
|
"key-aliases={} key-same-pages={} key-empty-slots={} key-full-sets={} "
|
||||||
|
"key-stale-translations={} empty-inserts={} replacements={} "
|
||||||
|
"icache-clears={} code-space-clears={} discarded-blocks={}",
|
||||||
|
m_core->GetJitCompiledBlockCount(), m_core->GetJitLifetimeCompiledBlockCount(),
|
||||||
|
m_core->GetJitBlockLookupCount(), m_core->GetJitBlockCacheHitCount(),
|
||||||
|
m_core->GetJitDispatchSlowCount(), m_core->GetJitDispatchStaleGenerationCount(),
|
||||||
|
m_core->GetJitDispatchPhysicalAliasCount(), m_core->GetJitDispatchCollisionCount(),
|
||||||
|
m_core->GetJitDispatchKeyMissCount(), m_core->GetJitDispatchPageMissCount(),
|
||||||
|
m_core->GetJitDispatchEmptyEntryCount(),
|
||||||
|
m_core->GetJitDispatchGeneratedKeyMismatchCount(),
|
||||||
|
m_core->GetJitDispatchGeneratedSetMismatchCount(),
|
||||||
|
m_core->GetJitDispatchKeyMissWithExactMatchCount(),
|
||||||
|
m_core->GetJitDispatchKeyMissWithPhysicalAliasCount(),
|
||||||
|
m_core->GetJitDispatchKeyMissWithSamePhysicalPageCount(),
|
||||||
|
m_core->GetJitDispatchKeyMissWithEmptySlotCount(),
|
||||||
|
m_core->GetJitDispatchKeyMissWithFullSetCount(),
|
||||||
|
m_core->GetJitDispatchKeyMissWithStaleTranslationCount(),
|
||||||
|
m_core->GetJitFastEntryEmptyInsertCount(),
|
||||||
|
m_core->GetJitFastEntryReplacementCount(),
|
||||||
|
m_core->GetJitInstructionCacheClearCount(), m_core->GetJitCodeSpaceClearCount(),
|
||||||
|
m_core->GetJitDiscardedBlockCount());
|
||||||
std::array<u32, 4> hot_sram_read_pages{};
|
std::array<u32, 4> hot_sram_read_pages{};
|
||||||
std::array<u64, 4> hot_sram_read_page_counts{};
|
std::array<u64, 4> hot_sram_read_page_counts{};
|
||||||
std::array<u32, 4> hot_sram_write_pages{};
|
std::array<u32, 4> hot_sram_write_pages{};
|
||||||
@@ -335,6 +361,12 @@ void Starlet::RunSlice(s64 cycles_late)
|
|||||||
hot_sram_write_page_counts[1], hot_sram_write_pages[2],
|
hot_sram_write_page_counts[1], hot_sram_write_pages[2],
|
||||||
hot_sram_write_page_counts[2], hot_sram_write_pages[3],
|
hot_sram_write_page_counts[2], hot_sram_write_pages[3],
|
||||||
hot_sram_write_page_counts[3]);
|
hot_sram_write_page_counts[3]);
|
||||||
|
for (const auto& [key, samples] : m_core->TakeHotJitSlowMemorySamples(8))
|
||||||
|
{
|
||||||
|
const bool write = (key >> 32) != 0;
|
||||||
|
INFO_LOG_FMT(IOS, "Starlet slow memory {} address={:#010x} samples={}",
|
||||||
|
write ? "write" : "read", static_cast<u32>(key), samples);
|
||||||
|
}
|
||||||
const ARMCore::HotPCSample current = m_core->GetCurrentPCSample();
|
const ARMCore::HotPCSample current = m_core->GetCurrentPCSample();
|
||||||
INFO_LOG_FMT(IOS, "Starlet current PC {:#010x} {} instruction={:#010x}", current.address,
|
INFO_LOG_FMT(IOS, "Starlet current PC {:#010x} {} instruction={:#010x}", current.address,
|
||||||
current.thumb ? "Thumb" : "ARM", current.instruction);
|
current.thumb ? "Thumb" : "ARM", current.instruction);
|
||||||
|
|||||||
@@ -189,13 +189,6 @@ constexpr u32 OHCI_PORT_CHANGE_MASK = 0x001f0000;
|
|||||||
constexpr u32 OHCI_FRAME_CYCLES = 243000;
|
constexpr u32 OHCI_FRAME_CYCLES = 243000;
|
||||||
constexpr u16 OHCI1_ATTACH_DELAY_FRAMES = 100;
|
constexpr u16 OHCI1_ATTACH_DELAY_FRAMES = 100;
|
||||||
constexpr u64 WIIMOTE_UPDATE_CYCLES = 243000000 / Wiimote::UPDATE_FREQ;
|
constexpr u64 WIIMOTE_UPDATE_CYCLES = 243000000 / Wiimote::UPDATE_FREQ;
|
||||||
// Poll host controls at Dolphin's normal 200 Hz, but do not wake the
|
|
||||||
// interpreted IOS Bluetooth stack for every poll. A 30 Hz steady HID stream
|
|
||||||
// leaves substantially more host time for Broadway; button transitions bypass
|
|
||||||
// this throttle below so presses and releases still reach IOS promptly.
|
|
||||||
constexpr u32 WIIMOTE_REPORT_FREQUENCY = 30;
|
|
||||||
constexpr u64 WIIMOTE_REPORT_CYCLES = 243000000 / WIIMOTE_REPORT_FREQUENCY;
|
|
||||||
static_assert(WIIMOTE_REPORT_FREQUENCY <= Wiimote::UPDATE_FREQ);
|
|
||||||
// Hollywood completes the internal OHCI1 port reset before IOS's first 2 ms
|
// Hollywood completes the internal OHCI1 port reset before IOS's first 2 ms
|
||||||
// poll. Using the generic 10 ms upper-bound timing leaves IOS with RHSC masked
|
// poll. Using the generic 10 ms upper-bound timing leaves IOS with RHSC masked
|
||||||
// when PRSC arrives.
|
// when PRSC arrives.
|
||||||
@@ -334,6 +327,7 @@ constexpr u32 HW_USBFRCRST = HW_BASE + 0x88;
|
|||||||
constexpr u32 HW_SRNPROT = HW_BASE + 0x60;
|
constexpr u32 HW_SRNPROT = HW_BASE + 0x60;
|
||||||
constexpr u32 HW_AHBPROT = HW_BASE + 0x64;
|
constexpr u32 HW_AHBPROT = HW_BASE + 0x64;
|
||||||
constexpr u32 HW_TIMER = HW_BASE + 0x10;
|
constexpr u32 HW_TIMER = HW_BASE + 0x10;
|
||||||
|
constexpr u64 TIMER_CLOCK_DIVISOR = 128;
|
||||||
constexpr u32 HW_ALARM = HW_BASE + 0x14;
|
constexpr u32 HW_ALARM = HW_BASE + 0x14;
|
||||||
constexpr u32 HW_GPIO_ENABLE = HW_BASE + 0xdc;
|
constexpr u32 HW_GPIO_ENABLE = HW_BASE + 0xdc;
|
||||||
constexpr u32 HW_GPIO_OUT = HW_BASE + 0xe0;
|
constexpr u32 HW_GPIO_OUT = HW_BASE + 0xe0;
|
||||||
@@ -683,6 +677,7 @@ void StarletMemory::Reset()
|
|||||||
for (size_t controller = 0; controller < OHCI_BASES.size(); ++controller)
|
for (size_t controller = 0; controller < OHCI_BASES.size(); ++controller)
|
||||||
ResetOHCIController(controller);
|
ResetOHCIController(controller);
|
||||||
m_arm_cycles = 0;
|
m_arm_cycles = 0;
|
||||||
|
m_timer_offset = 0;
|
||||||
m_boot0_mapped = true;
|
m_boot0_mapped = true;
|
||||||
m_sram_split_mode = false;
|
m_sram_split_mode = false;
|
||||||
// Retail Hollywood production revision. Early firmware branches on this
|
// Retail Hollywood production revision. Early firmware branches on this
|
||||||
@@ -786,6 +781,7 @@ void StarletMemory::DoState(PointerWrap& p)
|
|||||||
p.Do(m_seeprom_write_pending);
|
p.Do(m_seeprom_write_pending);
|
||||||
p.Do(m_seeprom_write_all);
|
p.Do(m_seeprom_write_all);
|
||||||
p.Do(m_arm_cycles);
|
p.Do(m_arm_cycles);
|
||||||
|
p.Do(m_timer_offset);
|
||||||
p.Do(m_initialized);
|
p.Do(m_initialized);
|
||||||
p.Do(m_boot0_mapped);
|
p.Do(m_boot0_mapped);
|
||||||
p.Do(m_sram_split_mode);
|
p.Do(m_sram_split_mode);
|
||||||
@@ -799,10 +795,11 @@ void StarletMemory::DoState(PointerWrap& p)
|
|||||||
|
|
||||||
u32 StarletMemory::GetTimer() const
|
u32 StarletMemory::GetTimer() const
|
||||||
{
|
{
|
||||||
// Hollywood's 19.2 MHz timer is clocked at 32/405 of the 243 MHz Starlet
|
// Hollywood's timer runs at Starlet / 128 (1.8984375 MHz at 243 MHz).
|
||||||
// clock.
|
// Keep the divider phase across scheduler slices. The writable counter has
|
||||||
const u64 timer = (m_arm_cycles / 405) * 32 + ((m_arm_cycles % 405) * 32) / 405;
|
// its own offset so IOS resetting HW_TIMER cannot rewind peripheral time.
|
||||||
return static_cast<u32>(timer);
|
// https://wiibrew.org/wiki/Hardware/Starlet_Timer
|
||||||
|
return static_cast<u32>(m_arm_cycles / TIMER_CLOCK_DIVISOR) + m_timer_offset;
|
||||||
}
|
}
|
||||||
|
|
||||||
std::optional<u32> StarletMemory::TryReadBroadwayResetInstruction(u32 address) const
|
std::optional<u32> StarletMemory::TryReadBroadwayResetInstruction(u32 address) const
|
||||||
@@ -895,6 +892,32 @@ const bool* StarletMemory::GetFastmemSRAMSplitMode() const
|
|||||||
return &m_sram_split_mode;
|
return &m_sram_split_mode;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
const u8* StarletMemory::GetDirectMemoryPointer(u32 address, u32 size) const
|
||||||
|
{
|
||||||
|
if (size == 0 || address > std::numeric_limits<u32>::max() - (size - 1))
|
||||||
|
return nullptr;
|
||||||
|
|
||||||
|
const u32 last_address = address + size - 1;
|
||||||
|
if (IsMemoryAddress(address) && IsMemoryAddress(last_address))
|
||||||
|
{
|
||||||
|
if (u8* const base = m_system.GetMemory().GetPhysicalBase())
|
||||||
|
return base + address;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Page tables can live in Hollywood SRAM. Respect the same boot0 overlay and A/B split mapping
|
||||||
|
// as normal bus reads, and only expose a pointer when the whole descriptor is contiguous.
|
||||||
|
if (IsSRAMWindowAddress(address) && IsSRAMWindowAddress(last_address) &&
|
||||||
|
!IsBootROMAddress(address) && !IsBootROMAddress(last_address))
|
||||||
|
{
|
||||||
|
const u32 offset = GetSRAMOffset(address);
|
||||||
|
const u32 end_offset = GetSRAMOffset(last_address);
|
||||||
|
if (offset != INVALID_SRAM_OFFSET && end_offset == offset + size - 1)
|
||||||
|
return m_sram.data() + offset;
|
||||||
|
}
|
||||||
|
|
||||||
|
return nullptr;
|
||||||
|
}
|
||||||
|
|
||||||
bool StarletMemory::IsMemoryAddress(u32 address)
|
bool StarletMemory::IsMemoryAddress(u32 address)
|
||||||
{
|
{
|
||||||
return address < Memory::MEM1_SIZE_RETAIL ||
|
return address < Memory::MEM1_SIZE_RETAIL ||
|
||||||
@@ -1677,25 +1700,18 @@ void StarletMemory::UpdateWiimotes()
|
|||||||
next_calls[i] = m_wiimotes[i]->PrepareInput(&states[i]);
|
next_calls[i] = m_wiimotes[i]->PrepareInput(&states[i]);
|
||||||
}
|
}
|
||||||
|
|
||||||
const u64 previous_update_cycles =
|
// Deliver input at the normal Wii Remote cadence, even when buttons have not
|
||||||
m_arm_cycles >= WIIMOTE_UPDATE_CYCLES ? m_arm_cycles - WIIMOTE_UPDATE_CYCLES : 0;
|
// changed. KPAD repeat timing depends on fresh reports; starving KPADRead can
|
||||||
const bool report_due =
|
// also leave a previous trigger visible across successive menu frames.
|
||||||
m_arm_cycles / WIIMOTE_REPORT_CYCLES != previous_update_cycles / WIIMOTE_REPORT_CYCLES;
|
|
||||||
|
|
||||||
for (size_t i = 0; i < m_wiimotes.size(); ++i)
|
for (size_t i = 0; i < m_wiimotes.size(); ++i)
|
||||||
{
|
{
|
||||||
if (!m_wiimotes[i])
|
if (!m_wiimotes[i])
|
||||||
continue;
|
continue;
|
||||||
|
|
||||||
const bool button_changed = states[i].buttons.hex != m_last_wiimote_buttons[i];
|
|
||||||
if (next_calls[i] != IOS::HLE::WiimoteDevice::NextUpdateInputCall::Update || report_due ||
|
|
||||||
button_changed)
|
|
||||||
{
|
|
||||||
m_wiimotes[i]->UpdateInput(next_calls[i], states[i]);
|
m_wiimotes[i]->UpdateInput(next_calls[i], states[i]);
|
||||||
if (next_calls[i] == IOS::HLE::WiimoteDevice::NextUpdateInputCall::Update)
|
if (next_calls[i] == IOS::HLE::WiimoteDevice::NextUpdateInputCall::Update)
|
||||||
m_last_wiimote_buttons[i] = states[i].buttons.hex;
|
m_last_wiimote_buttons[i] = states[i].buttons.hex;
|
||||||
}
|
}
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|
||||||
void StarletMemory::SendACLPacket(const bdaddr_t& source, const u8* data, u32 size)
|
void StarletMemory::SendACLPacket(const bdaddr_t& source, const u8* data, u32 size)
|
||||||
@@ -3326,6 +3342,22 @@ u16 StarletMemory::Read16(u32 address)
|
|||||||
|
|
||||||
u32 StarletMemory::Read32(u32 address)
|
u32 StarletMemory::Read32(u32 address)
|
||||||
{
|
{
|
||||||
|
// ARM IOS accesses the Hollywood timer and interrupt controller almost exclusively as aligned
|
||||||
|
// words. Falling back to ARMBus::Read32 decomposes each access into four virtual Read8 calls;
|
||||||
|
// every byte then repeats the complete device-range decoder and IPC register switch. Preserve
|
||||||
|
// exactly the same live values while resolving these two hottest register families once.
|
||||||
|
if ((address & 3) == 0)
|
||||||
|
{
|
||||||
|
if (address == HW_TIMER)
|
||||||
|
return GetTimer();
|
||||||
|
if ((address >= HW_BASE && address <= HW_BASE + 0x0c) ||
|
||||||
|
(address >= HW_BASE + 0x30 && address <= HW_BASE + 0x40) || address == HW_AHBPROT ||
|
||||||
|
address == HW_RESETS)
|
||||||
|
{
|
||||||
|
return m_system.GetWiiIPC().ReadStarletRegister(address - HW_BASE);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
if ((address & 3) == 0 && IsDIAddress(address))
|
if ((address & 3) == 0 && IsDIAddress(address))
|
||||||
{
|
{
|
||||||
MMIO::Mapping* const mmio = m_system.GetMemory().GetMMIOMapping();
|
MMIO::Mapping* const mmio = m_system.GetMemory().GetMMIOMapping();
|
||||||
@@ -3513,7 +3545,7 @@ void StarletMemory::Write8(u32 address, u8 value)
|
|||||||
else if (word_address == HW_OTPCMD)
|
else if (word_address == HW_OTPCMD)
|
||||||
HandleOTPCommand(word);
|
HandleOTPCommand(word);
|
||||||
else if (word_address == HW_TIMER)
|
else if (word_address == HW_TIMER)
|
||||||
m_arm_cycles = static_cast<u64>(word) * 405 / 32;
|
m_timer_offset = word - static_cast<u32>(m_arm_cycles / TIMER_CLOCK_DIVISOR);
|
||||||
else if (word_address == HW_ALARM)
|
else if (word_address == HW_ALARM)
|
||||||
{
|
{
|
||||||
// HW_ALARM is a comparator, not an interrupt acknowledgement register.
|
// HW_ALARM is a comparator, not an interrupt acknowledgement register.
|
||||||
@@ -3602,6 +3634,29 @@ void StarletMemory::Write16(u32 address, u16 value)
|
|||||||
|
|
||||||
void StarletMemory::Write32(u32 address, u32 value)
|
void StarletMemory::Write32(u32 address, u32 value)
|
||||||
{
|
{
|
||||||
|
// Match the aligned read fast path above. WriteRegister keeps the byte-addressable backing image
|
||||||
|
// coherent, while the exact device handlers retain W1C interrupt flags, reset side effects and
|
||||||
|
// the Starlet/Broadway scheduling boundary of the four-byte Write8 path.
|
||||||
|
if ((address & 3) == 0 && address == HW_TIMER)
|
||||||
|
{
|
||||||
|
WriteRegister(address, value);
|
||||||
|
m_timer_offset = value - static_cast<u32>(m_arm_cycles / TIMER_CLOCK_DIVISOR);
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
if ((address & 3) == 0 && ((address >= HW_BASE && address <= HW_BASE + 0x0c) ||
|
||||||
|
(address >= HW_BASE + 0x30 && address <= HW_BASE + 0x40) ||
|
||||||
|
address == HW_AHBPROT || address == HW_RESETS))
|
||||||
|
{
|
||||||
|
WriteRegister(address, value);
|
||||||
|
m_system.GetWiiIPC().WriteStarletRegister(address - HW_BASE, value);
|
||||||
|
if (address == HW_BASE + 0x0c && (value & 0x09) != 0)
|
||||||
|
{
|
||||||
|
if (Starlet* const starlet = m_system.GetStarlet())
|
||||||
|
starlet->YieldForIPC();
|
||||||
|
}
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
if ((address & 3) == 0 && IsDIAddress(address))
|
if ((address & 3) == 0 && IsDIAddress(address))
|
||||||
{
|
{
|
||||||
MMIO::Mapping* const mmio = m_system.GetMemory().GetMMIOMapping();
|
MMIO::Mapping* const mmio = m_system.GetMemory().GetMMIOMapping();
|
||||||
|
|||||||
@@ -69,6 +69,7 @@ public:
|
|||||||
u8* GetFastmemSRAMBase() const override;
|
u8* GetFastmemSRAMBase() const override;
|
||||||
const bool* GetFastmemBoot0Mapped() const override;
|
const bool* GetFastmemBoot0Mapped() const override;
|
||||||
const bool* GetFastmemSRAMSplitMode() const override;
|
const bool* GetFastmemSRAMSplitMode() const override;
|
||||||
|
const u8* GetDirectMemoryPointer(u32 address, u32 size) const override;
|
||||||
|
|
||||||
u64 GetCycles() const { return m_arm_cycles; }
|
u64 GetCycles() const { return m_arm_cycles; }
|
||||||
std::optional<u32> TryReadBroadwayResetInstruction(u32 address) const;
|
std::optional<u32> TryReadBroadwayResetInstruction(u32 address) const;
|
||||||
@@ -302,6 +303,7 @@ private:
|
|||||||
bool m_seeprom_write_pending = false;
|
bool m_seeprom_write_pending = false;
|
||||||
bool m_seeprom_write_all = false;
|
bool m_seeprom_write_all = false;
|
||||||
u64 m_arm_cycles = 0;
|
u64 m_arm_cycles = 0;
|
||||||
|
u32 m_timer_offset = 0;
|
||||||
bool m_initialized = false;
|
bool m_initialized = false;
|
||||||
bool m_boot0_mapped = true;
|
bool m_boot0_mapped = true;
|
||||||
bool m_sram_split_mode = false;
|
bool m_sram_split_mode = false;
|
||||||
|
|||||||
@@ -3,7 +3,11 @@
|
|||||||
|
|
||||||
#include "Core/PowerPC/Jit64/Jit.h"
|
#include "Core/PowerPC/Jit64/Jit.h"
|
||||||
|
|
||||||
|
#include <algorithm>
|
||||||
|
#include <chrono>
|
||||||
|
#include <cstdlib>
|
||||||
#include <map>
|
#include <map>
|
||||||
|
#include <optional>
|
||||||
#include <span>
|
#include <span>
|
||||||
#include <sstream>
|
#include <sstream>
|
||||||
#include <string>
|
#include <string>
|
||||||
@@ -18,6 +22,7 @@
|
|||||||
#endif
|
#endif
|
||||||
|
|
||||||
#include "Common/CommonTypes.h"
|
#include "Common/CommonTypes.h"
|
||||||
|
#include "Common/FileUtil.h"
|
||||||
#include "Common/GekkoDisassembler.h"
|
#include "Common/GekkoDisassembler.h"
|
||||||
#include "Common/HostDisassembler.h"
|
#include "Common/HostDisassembler.h"
|
||||||
#include "Common/IOFile.h"
|
#include "Common/IOFile.h"
|
||||||
@@ -51,6 +56,111 @@
|
|||||||
using namespace Gen;
|
using namespace Gen;
|
||||||
using namespace PowerPC;
|
using namespace PowerPC;
|
||||||
|
|
||||||
|
namespace
|
||||||
|
{
|
||||||
|
// Temporary, opt-in event diagnostics. Addresses and expected instructions come from a local
|
||||||
|
// tracepoint file, not from a title-specific emulation rule. No guest state is modified.
|
||||||
|
struct PPCEventTrace
|
||||||
|
{
|
||||||
|
std::map<u32, u32> points;
|
||||||
|
File::IOFile output;
|
||||||
|
std::chrono::steady_clock::time_point start;
|
||||||
|
std::chrono::steady_clock::time_point last_flush;
|
||||||
|
u32 rows = 0;
|
||||||
|
|
||||||
|
void Init()
|
||||||
|
{
|
||||||
|
points.clear();
|
||||||
|
rows = 0;
|
||||||
|
const char* config_path = std::getenv("DOLPHIN_PPC_EVENT_TRACE");
|
||||||
|
const char* output_path = std::getenv("DOLPHIN_PPC_EVENT_TRACE_OUTPUT");
|
||||||
|
if (!config_path || !output_path)
|
||||||
|
return;
|
||||||
|
std::string config;
|
||||||
|
if (!File::ReadFileToString(config_path, config))
|
||||||
|
return;
|
||||||
|
std::istringstream input(config);
|
||||||
|
u32 pc, instruction;
|
||||||
|
while (input >> std::hex >> pc >> instruction)
|
||||||
|
{
|
||||||
|
if ((pc & 3) || points.size() >= 32)
|
||||||
|
{
|
||||||
|
points.clear();
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
points.emplace(pc, instruction);
|
||||||
|
}
|
||||||
|
if (points.empty() || !output.Open(output_path, "ab", File::SharedAccess::Read))
|
||||||
|
{
|
||||||
|
points.clear();
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
output.WriteString("seq,wall_us,ticks,pc,lr,ctr,r3,r4,r5,r6,r7,r8,"
|
||||||
|
"mem0,mem1,mem2,mem3,mem4,mem5,mem6\n");
|
||||||
|
output.Flush();
|
||||||
|
start = last_flush = std::chrono::steady_clock::now();
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
PPCEventTrace s_ppc_event_trace;
|
||||||
|
|
||||||
|
// Inspect only the standard cached MEM1/MEM2 aliases. In particular, don't use an MMU load here:
|
||||||
|
// it could populate a cache line or change PLRU just by observing the packet.
|
||||||
|
std::optional<u32> PeekEventWord(Core::System& system, u32 address)
|
||||||
|
{
|
||||||
|
if ((address & 3) || (address >> 28 != 8 && address >> 28 != 9))
|
||||||
|
return std::nullopt;
|
||||||
|
const u32 physical = address & 0x3fffffff;
|
||||||
|
auto& memory = system.GetMemory();
|
||||||
|
if (!(physical < memory.GetRamSizeReal() && memory.GetRamSizeReal() - physical >= 4) &&
|
||||||
|
!(physical >= 0x10000000 && physical - 0x10000000 < memory.GetExRamSizeReal() &&
|
||||||
|
memory.GetExRamSizeReal() - (physical - 0x10000000) >= 4))
|
||||||
|
return std::nullopt;
|
||||||
|
const auto& ppc = system.GetPowerPC().GetPPCState();
|
||||||
|
const UReg_HID0 hid0{.Hex = ppc.spr[SPR_HID0]};
|
||||||
|
if (ppc.m_enable_dcache && hid0.DCE)
|
||||||
|
{
|
||||||
|
const u32 set = (physical >> 5) & 127;
|
||||||
|
const auto& cache = ppc.dCache;
|
||||||
|
for (u32 way = 0; way < 8; ++way)
|
||||||
|
{
|
||||||
|
if ((cache.valid[set] & (1 << way)) && cache.addrs[set][way] == (physical & ~31U))
|
||||||
|
return Common::swap32(cache.data[set][way][(physical & 31) / 4]);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
const u8* data = memory.GetPointerForRange(physical, 4);
|
||||||
|
return data ? std::optional<u32>(Common::swap32(data)) : std::nullopt;
|
||||||
|
}
|
||||||
|
|
||||||
|
void RecordPPCEvent(Core::System* system, u32 pc)
|
||||||
|
{
|
||||||
|
auto& trace = s_ppc_event_trace;
|
||||||
|
if (!trace.output.IsOpen())
|
||||||
|
return;
|
||||||
|
const auto now = std::chrono::steady_clock::now();
|
||||||
|
const auto& ppc = system->GetPowerPC().GetPPCState();
|
||||||
|
std::string row =
|
||||||
|
fmt::format("{},{},{},{:08x},{:08x},{:08x}", ++trace.rows,
|
||||||
|
std::chrono::duration_cast<std::chrono::microseconds>(now - trace.start).count(),
|
||||||
|
system->GetCoreTiming().GetTicks(), pc, LR(ppc), CTR(ppc));
|
||||||
|
for (u32 reg = 3; reg <= 8; ++reg)
|
||||||
|
row += fmt::format(",{:08x}", ppc.gpr[reg]);
|
||||||
|
for (u32 word = 0; word < 7; ++word)
|
||||||
|
{
|
||||||
|
const auto value = PeekEventWord(*system, ppc.gpr[4] + word * 4);
|
||||||
|
row += value ? fmt::format(",{:08x}", *value) : ",NA";
|
||||||
|
}
|
||||||
|
row += '\n';
|
||||||
|
if (!trace.output.WriteString(row) || trace.rows >= 300000)
|
||||||
|
trace.output.Close();
|
||||||
|
else if (now - trace.last_flush >= std::chrono::milliseconds(250))
|
||||||
|
{
|
||||||
|
trace.output.Flush();
|
||||||
|
trace.last_flush = now;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
} // namespace
|
||||||
|
|
||||||
// Dolphin's PowerPC->x86_64 JIT dynamic recompiler
|
// Dolphin's PowerPC->x86_64 JIT dynamic recompiler
|
||||||
// Written mostly by ector (hrydgard)
|
// Written mostly by ector (hrydgard)
|
||||||
// Features:
|
// Features:
|
||||||
@@ -255,6 +365,7 @@ bool Jit64::BackPatch(SContext* ctx)
|
|||||||
|
|
||||||
void Jit64::Init()
|
void Jit64::Init()
|
||||||
{
|
{
|
||||||
|
s_ppc_event_trace.Init();
|
||||||
InitFastmemArena();
|
InitFastmemArena();
|
||||||
|
|
||||||
RefreshConfig();
|
RefreshConfig();
|
||||||
@@ -338,6 +449,9 @@ void Jit64::ResetFreeMemoryRanges()
|
|||||||
|
|
||||||
void Jit64::Shutdown()
|
void Jit64::Shutdown()
|
||||||
{
|
{
|
||||||
|
if (s_ppc_event_trace.output.IsOpen())
|
||||||
|
s_ppc_event_trace.output.Close();
|
||||||
|
s_ppc_event_trace.points.clear();
|
||||||
FreeCodeSpace();
|
FreeCodeSpace();
|
||||||
|
|
||||||
auto& memory = m_system.GetMemory();
|
auto& memory = m_system.GetMemory();
|
||||||
@@ -824,6 +938,19 @@ void Jit64::Jit(u32 em_address, bool clear_cache_and_retry_on_failure)
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
if (!s_ppc_event_trace.points.empty())
|
||||||
|
{
|
||||||
|
// Force tracepoints to be block entries, where all guest registers are materialized. Avoid
|
||||||
|
// following branches past these boundaries. Ordinary runs retain their normal optimizations.
|
||||||
|
analyzer.ClearOption(PPCAnalyst::PPCAnalyzer::OPTION_BRANCH_FOLLOW);
|
||||||
|
const auto next = s_ppc_event_trace.points.lower_bound(em_address);
|
||||||
|
if (next != s_ppc_event_trace.points.end())
|
||||||
|
{
|
||||||
|
const size_t distance = (next->first - em_address) / 4;
|
||||||
|
block_size = std::min(block_size, distance == 0 ? size_t{1} : distance);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// Analyze the block, collect all instructions it is made of (including inlining,
|
// Analyze the block, collect all instructions it is made of (including inlining,
|
||||||
// if that is enabled), reorder instructions for optimal performance, and join joinable
|
// if that is enabled), reorder instructions for optimal performance, and join joinable
|
||||||
// instructions.
|
// instructions.
|
||||||
@@ -926,6 +1053,15 @@ bool Jit64::DoJit(u32 em_address, JitBlock* b, u32 nextPC)
|
|||||||
// TODO: Test if this or AlignCode16 make a difference from GetCodePtr
|
// TODO: Test if this or AlignCode16 make a difference from GetCodePtr
|
||||||
b->normalEntry = AlignCode4();
|
b->normalEntry = AlignCode4();
|
||||||
|
|
||||||
|
const auto tracepoint = s_ppc_event_trace.points.find(em_address);
|
||||||
|
if (tracepoint != s_ppc_event_trace.points.end() &&
|
||||||
|
m_code_buffer[0].inst.hex == tracepoint->second)
|
||||||
|
{
|
||||||
|
ABI_PushRegistersAndAdjustStack({}, 0);
|
||||||
|
ABI_CallFunctionPC(RecordPPCEvent, &m_system, em_address);
|
||||||
|
ABI_PopRegistersAndAdjustStack({}, 0);
|
||||||
|
}
|
||||||
|
|
||||||
// Used to get a trace of the last few blocks before a crash, sometimes VERY useful
|
// Used to get a trace of the last few blocks before a crash, sometimes VERY useful
|
||||||
if (m_im_here_debug)
|
if (m_im_here_debug)
|
||||||
{
|
{
|
||||||
|
|||||||
@@ -262,6 +262,10 @@ void Jit64AsmRoutineManager::GenerateCommon()
|
|||||||
GenFres();
|
GenFres();
|
||||||
mfcr = AlignCode4();
|
mfcr = AlignCode4();
|
||||||
GenMfcr();
|
GenMfcr();
|
||||||
|
dcache32_read_hit_dbat = AlignCode4();
|
||||||
|
GenDCache32Hit(false);
|
||||||
|
dcache32_write_hit_dbat = AlignCode4();
|
||||||
|
GenDCache32Hit(true);
|
||||||
cdts = AlignCode4();
|
cdts = AlignCode4();
|
||||||
GenConvertDoubleToSingle();
|
GenConvertDoubleToSingle();
|
||||||
fmadds_eft = AlignCode4();
|
fmadds_eft = AlignCode4();
|
||||||
|
|||||||
@@ -376,9 +376,53 @@ void EmuCodeBlock::SafeLoadToReg(X64Reg reg_value, const Gen::OpArg& opAddress,
|
|||||||
LEA(32, RSCRATCH, MDisp(opAddress.GetSimpleReg(), offset));
|
LEA(32, RSCRATCH, MDisp(opAddress.GetSimpleReg(), offset));
|
||||||
}
|
}
|
||||||
|
|
||||||
FixupBranch exit;
|
|
||||||
const bool dr_set =
|
const bool dr_set =
|
||||||
(flags & SAFE_LOADSTORE_DR_ON) || (m_jit.m_ppc_state.feature_flags & FEATURE_FLAG_MSR_DR);
|
(flags & SAFE_LOADSTORE_DR_ON) || (m_jit.m_ppc_state.feature_flags & FEATURE_FLAG_MSR_DR);
|
||||||
|
FixupBranch accurate_dcache_done;
|
||||||
|
const bool accurate_dcache_hit_path =
|
||||||
|
!force_slow_access && dr_set && accessSize == 32 && !m_jit.jo.memcheck &&
|
||||||
|
m_jit.m_ppc_state.m_enable_dcache &&
|
||||||
|
(reg_addr != ABI_RETURN || (offset && opAddress.GetSimpleReg() != ABI_RETURN));
|
||||||
|
if (accurate_dcache_hit_path)
|
||||||
|
{
|
||||||
|
BitSet32 fast_registers_in_use = registersInUse;
|
||||||
|
if (reg_addr != ABI_RETURN)
|
||||||
|
fast_registers_in_use[reg_addr] = true;
|
||||||
|
else
|
||||||
|
fast_registers_in_use[opAddress.GetSimpleReg()] = true;
|
||||||
|
|
||||||
|
// Preserve scratch registers that remain live, including the original effective address
|
||||||
|
// needed by a cache-miss fallback. RDX can hold an address or paired-load intermediate even
|
||||||
|
// though the GPR allocator does not use it.
|
||||||
|
const bool preserve_rdx = fast_registers_in_use[RSCRATCH2];
|
||||||
|
const bool preserve_rcx = fast_registers_in_use[RSCRATCH_EXTRA];
|
||||||
|
if (preserve_rdx)
|
||||||
|
PUSH(RSCRATCH2);
|
||||||
|
if (preserve_rcx)
|
||||||
|
PUSH(RSCRATCH_EXTRA);
|
||||||
|
if (reg_addr != RSCRATCH)
|
||||||
|
MOV(32, R(RSCRATCH), R(reg_addr));
|
||||||
|
CALL(CommonAsmRoutines::dcache32_read_hit_dbat);
|
||||||
|
if (preserve_rcx)
|
||||||
|
POP(RSCRATCH_EXTRA);
|
||||||
|
if (preserve_rdx)
|
||||||
|
POP(RSCRATCH2);
|
||||||
|
|
||||||
|
TEST(64, R(ABI_RETURN), R(ABI_RETURN));
|
||||||
|
const FixupBranch slow = J_CC(CC_Z, Jump::Near);
|
||||||
|
SUB(64, R(ABI_RETURN), Imm8(1));
|
||||||
|
if (reg_value != ABI_RETURN)
|
||||||
|
MOV(32, R(reg_value), R(ABI_RETURN));
|
||||||
|
accurate_dcache_done = J(Jump::Near);
|
||||||
|
SetJumpTarget(slow);
|
||||||
|
|
||||||
|
// An offset address lives in RAX, which is also the ABI return register. Recreate it only on
|
||||||
|
// the miss path before calling the full helper.
|
||||||
|
if (reg_addr == ABI_RETURN && offset)
|
||||||
|
LEA(32, RSCRATCH, MDisp(opAddress.GetSimpleReg(), offset));
|
||||||
|
}
|
||||||
|
|
||||||
|
FixupBranch exit;
|
||||||
const bool fast_check_address =
|
const bool fast_check_address =
|
||||||
!force_slow_access && dr_set && m_jit.jo.fastmem_arena && !m_jit.m_ppc_state.m_enable_dcache;
|
!force_slow_access && dr_set && m_jit.jo.fastmem_arena && !m_jit.m_ppc_state.m_enable_dcache;
|
||||||
if (fast_check_address)
|
if (fast_check_address)
|
||||||
@@ -438,6 +482,9 @@ void EmuCodeBlock::SafeLoadToReg(X64Reg reg_value, const Gen::OpArg& opAddress,
|
|||||||
}
|
}
|
||||||
SetJumpTarget(exit);
|
SetJumpTarget(exit);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
if (accurate_dcache_hit_path)
|
||||||
|
SetJumpTarget(accurate_dcache_done);
|
||||||
}
|
}
|
||||||
|
|
||||||
void EmuCodeBlock::SafeLoadToRegImmediate(X64Reg reg_value, u32 address, int accessSize,
|
void EmuCodeBlock::SafeLoadToRegImmediate(X64Reg reg_value, u32 address, int accessSize,
|
||||||
@@ -533,12 +580,15 @@ void EmuCodeBlock::SafeWriteRegToReg(OpArg reg_value, X64Reg reg_addr, int acces
|
|||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
const X64Reg original_reg_addr = reg_addr;
|
||||||
|
bool address_in_return_register = false;
|
||||||
if (offset)
|
if (offset)
|
||||||
{
|
{
|
||||||
if (flags & SAFE_LOADSTORE_CLOBBER_RSCRATCH_INSTEAD_OF_ADDR)
|
if (flags & SAFE_LOADSTORE_CLOBBER_RSCRATCH_INSTEAD_OF_ADDR)
|
||||||
{
|
{
|
||||||
LEA(32, RSCRATCH, MDisp(reg_addr, (u32)offset));
|
LEA(32, RSCRATCH, MDisp(reg_addr, (u32)offset));
|
||||||
reg_addr = RSCRATCH;
|
reg_addr = RSCRATCH;
|
||||||
|
address_in_return_register = true;
|
||||||
}
|
}
|
||||||
else
|
else
|
||||||
{
|
{
|
||||||
@@ -546,9 +596,53 @@ void EmuCodeBlock::SafeWriteRegToReg(OpArg reg_value, X64Reg reg_addr, int acces
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
FixupBranch exit;
|
|
||||||
const bool dr_set =
|
const bool dr_set =
|
||||||
(flags & SAFE_LOADSTORE_DR_ON) || (m_jit.m_ppc_state.feature_flags & FEATURE_FLAG_MSR_DR);
|
(flags & SAFE_LOADSTORE_DR_ON) || (m_jit.m_ppc_state.feature_flags & FEATURE_FLAG_MSR_DR);
|
||||||
|
FixupBranch accurate_dcache_done;
|
||||||
|
const bool accurate_dcache_hit_path =
|
||||||
|
!force_slow_access && dr_set && accessSize == 32 && swap && !m_jit.jo.memcheck &&
|
||||||
|
m_jit.m_ppc_state.m_enable_dcache &&
|
||||||
|
(reg_addr != ABI_RETURN || (address_in_return_register && original_reg_addr != ABI_RETURN)) &&
|
||||||
|
(!reg_value.IsSimpleReg() || reg_value.GetSimpleReg() != ABI_RETURN);
|
||||||
|
if (accurate_dcache_hit_path)
|
||||||
|
{
|
||||||
|
BitSet32 fast_registers_in_use = registersInUse;
|
||||||
|
if (reg_addr != ABI_RETURN)
|
||||||
|
fast_registers_in_use[reg_addr] = true;
|
||||||
|
else
|
||||||
|
fast_registers_in_use[original_reg_addr] = true;
|
||||||
|
if (reg_value.IsSimpleReg())
|
||||||
|
fast_registers_in_use[reg_value.GetSimpleReg()] = true;
|
||||||
|
|
||||||
|
// The value setup itself overwrites RDX, so save it before loading the call arguments.
|
||||||
|
// On a miss, the full MMU helper must see the original address and value registers.
|
||||||
|
const bool preserve_rdx = fast_registers_in_use[RSCRATCH2];
|
||||||
|
const bool preserve_rcx = fast_registers_in_use[RSCRATCH_EXTRA];
|
||||||
|
if (preserve_rdx)
|
||||||
|
PUSH(RSCRATCH2);
|
||||||
|
if (preserve_rcx)
|
||||||
|
PUSH(RSCRATCH_EXTRA);
|
||||||
|
// Move the address first because it is allowed to arrive in RDX, which is also the native
|
||||||
|
// routine's value input.
|
||||||
|
if (reg_addr != RSCRATCH)
|
||||||
|
MOV(32, R(RSCRATCH), R(reg_addr));
|
||||||
|
if (reg_value.IsImm())
|
||||||
|
MOV(32, R(RSCRATCH2), reg_value);
|
||||||
|
else if (reg_value.GetSimpleReg() != RSCRATCH2)
|
||||||
|
MOV(32, R(RSCRATCH2), reg_value);
|
||||||
|
CALL(CommonAsmRoutines::dcache32_write_hit_dbat);
|
||||||
|
if (preserve_rcx)
|
||||||
|
POP(RSCRATCH_EXTRA);
|
||||||
|
if (preserve_rdx)
|
||||||
|
POP(RSCRATCH2);
|
||||||
|
|
||||||
|
TEST(8, R(ABI_RETURN), R(ABI_RETURN));
|
||||||
|
accurate_dcache_done = J_CC(CC_NZ, Jump::Near);
|
||||||
|
if (address_in_return_register)
|
||||||
|
LEA(32, RSCRATCH, MDisp(original_reg_addr, (u32)offset));
|
||||||
|
}
|
||||||
|
|
||||||
|
FixupBranch exit;
|
||||||
const bool fast_check_address =
|
const bool fast_check_address =
|
||||||
!force_slow_access && dr_set && m_jit.jo.fastmem_arena && !m_jit.m_ppc_state.m_enable_dcache;
|
!force_slow_access && dr_set && m_jit.jo.fastmem_arena && !m_jit.m_ppc_state.m_enable_dcache;
|
||||||
if (fast_check_address)
|
if (fast_check_address)
|
||||||
@@ -615,6 +709,9 @@ void EmuCodeBlock::SafeWriteRegToReg(OpArg reg_value, X64Reg reg_addr, int acces
|
|||||||
}
|
}
|
||||||
SetJumpTarget(exit);
|
SetJumpTarget(exit);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
if (accurate_dcache_hit_path)
|
||||||
|
SetJumpTarget(accurate_dcache_done);
|
||||||
}
|
}
|
||||||
|
|
||||||
void EmuCodeBlock::SafeWriteRegToReg(Gen::X64Reg reg_value, Gen::X64Reg reg_addr, int accessSize,
|
void EmuCodeBlock::SafeWriteRegToReg(Gen::X64Reg reg_value, Gen::X64Reg reg_addr, int accessSize,
|
||||||
|
|||||||
@@ -14,8 +14,12 @@
|
|||||||
#include "Common/x64ABI.h"
|
#include "Common/x64ABI.h"
|
||||||
#include "Common/x64Emitter.h"
|
#include "Common/x64Emitter.h"
|
||||||
#include "Core/PowerPC/Gekko.h"
|
#include "Core/PowerPC/Gekko.h"
|
||||||
|
#include "Core/PowerPC/Jit64/Jit.h"
|
||||||
#include "Core/PowerPC/Jit64Common/Jit64Constants.h"
|
#include "Core/PowerPC/Jit64Common/Jit64Constants.h"
|
||||||
#include "Core/PowerPC/Jit64Common/Jit64PowerPCState.h"
|
#include "Core/PowerPC/Jit64Common/Jit64PowerPCState.h"
|
||||||
|
#include "Core/PowerPC/MMU.h"
|
||||||
|
#include "Core/PowerPC/PPCCache.h"
|
||||||
|
#include "Core/System.h"
|
||||||
|
|
||||||
#define QUANTIZED_REGS_TO_SAVE \
|
#define QUANTIZED_REGS_TO_SAVE \
|
||||||
(ABI_ALL_CALLER_SAVED & ~BitSet32{RSCRATCH, RSCRATCH2, RSCRATCH_EXTRA, XMM0 + 16, XMM1 + 16})
|
(ABI_ALL_CALLER_SAVED & ~BitSet32{RSCRATCH, RSCRATCH2, RSCRATCH_EXTRA, XMM0 + 16, XMM1 + 16})
|
||||||
@@ -24,11 +28,166 @@
|
|||||||
|
|
||||||
using namespace Gen;
|
using namespace Gen;
|
||||||
|
|
||||||
|
const u8* CommonAsmRoutines::dcache32_read_hit_dbat = nullptr;
|
||||||
|
const u8* CommonAsmRoutines::dcache32_write_hit_dbat = nullptr;
|
||||||
|
|
||||||
alignas(16) static const __m128i double_fraction = _mm_set_epi64x(0, 0x000fffffffffffff);
|
alignas(16) static const __m128i double_fraction = _mm_set_epi64x(0, 0x000fffffffffffff);
|
||||||
alignas(16) static const __m128i double_explicit_top_bit = _mm_set_epi64x(0, 0x0010000000000000);
|
alignas(16) static const __m128i double_explicit_top_bit = _mm_set_epi64x(0, 0x0010000000000000);
|
||||||
alignas(16) static const __m128i double_top_two_bits = _mm_set_epi64x(0, 0xc000000000000000);
|
alignas(16) static const __m128i double_top_two_bits = _mm_set_epi64x(0, 0xc000000000000000);
|
||||||
alignas(16) static const __m128i double_bottom_bits = _mm_set_epi64x(0, 0x07ffffffe0000000);
|
alignas(16) static const __m128i double_bottom_bits = _mm_set_epi64x(0, 0x07ffffffe0000000);
|
||||||
|
|
||||||
|
constexpr std::array<u8, 128 * PowerPC::CACHE_WAYS> s_dcache_plru_update = [] {
|
||||||
|
std::array<u8, 128 * PowerPC::CACHE_WAYS> result{};
|
||||||
|
for (u32 old_plru = 0; old_plru < 128; ++old_plru)
|
||||||
|
{
|
||||||
|
for (u32 way = 0; way < PowerPC::CACHE_WAYS; ++way)
|
||||||
|
{
|
||||||
|
result[old_plru * PowerPC::CACHE_WAYS + way] =
|
||||||
|
(old_plru & ~PowerPC::Cache::PLRU_MASK[way]) | PowerPC::Cache::PLRU_VALUE[way];
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return result;
|
||||||
|
}();
|
||||||
|
|
||||||
|
void CommonAsmRoutines::GenDCache32Hit(const bool write)
|
||||||
|
{
|
||||||
|
// This is the native equivalent of MMU::TryRead/WriteDCache32ForJit for the BAT-mapped
|
||||||
|
// MEM1/MEM2 hit case. Keeping it as a shared leaf routine avoids both the large C++ ABI
|
||||||
|
// register save and duplicating this sequence at every guest load/store.
|
||||||
|
//
|
||||||
|
// read: EAX = effective address; RAX = byte-swapped value + 1, or zero on miss
|
||||||
|
// write: EAX = effective address, EDX = guest value; EAX = one on hit, zero on miss
|
||||||
|
// Only the three Jit64 scratch registers are clobbered.
|
||||||
|
const void* const start = GetCodePtr();
|
||||||
|
auto& cache = m_jit.m_ppc_state.dCache;
|
||||||
|
auto& memory = m_jit.m_system.GetMemory();
|
||||||
|
ASSERT(!cache.lookup_table.empty());
|
||||||
|
ASSERT(!memory.GetEXRAM() || !cache.lookup_table_ex.empty());
|
||||||
|
|
||||||
|
// A write needs RDX for address translation, so keep its input value below the return address.
|
||||||
|
if (write)
|
||||||
|
PUSH(RSCRATCH2);
|
||||||
|
|
||||||
|
MOV(32, R(RSCRATCH2), R(RSCRATCH));
|
||||||
|
SHR(32, R(RSCRATCH2), Imm8(PowerPC::BAT_INDEX_SHIFT));
|
||||||
|
MOV(64, R(RSCRATCH_EXTRA), ImmPtr(m_jit.m_mmu.GetDBATTable().data()));
|
||||||
|
MOV(32, R(RSCRATCH2), MComplex(RSCRATCH_EXTRA, RSCRATCH2, SCALE_4, 0));
|
||||||
|
MOV(32, R(RSCRATCH_EXTRA), R(RSCRATCH2));
|
||||||
|
AND(32, R(RSCRATCH_EXTRA), Imm32(PowerPC::BAT_MAPPED_BIT | PowerPC::BAT_WI_BIT));
|
||||||
|
CMP(32, R(RSCRATCH_EXTRA), Imm32(PowerPC::BAT_MAPPED_BIT));
|
||||||
|
const FixupBranch invalid_bat = J_CC(CC_NE, Jump::Near);
|
||||||
|
|
||||||
|
AND(32, R(RSCRATCH2), Imm32(PowerPC::BAT_RESULT_MASK));
|
||||||
|
AND(32, R(RSCRATCH), Imm32(PowerPC::BAT_PAGE_SIZE - 1));
|
||||||
|
OR(32, R(RSCRATCH2), R(RSCRATCH));
|
||||||
|
|
||||||
|
// Select the lookup table while normalizing the physical address. Cache set and byte offset
|
||||||
|
// are identical for MEM1 and MEM2 after this normalization.
|
||||||
|
TEST(32, R(RSCRATCH2), Imm32(0xF8000000));
|
||||||
|
const FixupBranch mem1 = J_CC(CC_Z, Jump::Near);
|
||||||
|
|
||||||
|
MOV(32, R(RSCRATCH), R(RSCRATCH2));
|
||||||
|
AND(32, R(RSCRATCH), Imm32(0xF0000000));
|
||||||
|
CMP(32, R(RSCRATCH), Imm32(0x10000000));
|
||||||
|
const FixupBranch not_exram = J_CC(CC_NE, Jump::Near);
|
||||||
|
AND(32, R(RSCRATCH2), Imm32(0x0FFFFFFF));
|
||||||
|
CMP(32, R(RSCRATCH2), Imm32(memory.GetEXRAM() ? memory.GetExRamSizeReal() : 0));
|
||||||
|
const FixupBranch exram_out_of_range = J_CC(CC_AE, Jump::Near);
|
||||||
|
MOV(64, R(RSCRATCH), ImmPtr(cache.lookup_table_ex.data()));
|
||||||
|
const FixupBranch lookup_ready = J(Jump::Near);
|
||||||
|
|
||||||
|
SetJumpTarget(mem1);
|
||||||
|
AND(32, R(RSCRATCH2), Imm32(memory.GetRamMask()));
|
||||||
|
MOV(64, R(RSCRATCH), ImmPtr(cache.lookup_table.data()));
|
||||||
|
|
||||||
|
SetJumpTarget(lookup_ready);
|
||||||
|
MOV(32, R(RSCRATCH_EXTRA), R(RSCRATCH2));
|
||||||
|
SHR(32, R(RSCRATCH_EXTRA), Imm8(5));
|
||||||
|
MOVZX(32, 8, RSCRATCH_EXTRA, MComplex(RSCRATCH, RSCRATCH_EXTRA, SCALE_1, 0));
|
||||||
|
CMP(32, R(RSCRATCH_EXTRA), Imm32(0xff));
|
||||||
|
const FixupBranch cache_miss = J_CC(CC_E, Jump::Near);
|
||||||
|
|
||||||
|
// The generic helper preserves split cache-line transactions. Ordinary aligned 32-bit PPC
|
||||||
|
// traffic never takes this branch.
|
||||||
|
MOV(32, R(RSCRATCH), R(RSCRATCH2));
|
||||||
|
AND(32, R(RSCRATCH), Imm32(31));
|
||||||
|
CMP(32, R(RSCRATCH), Imm32(28));
|
||||||
|
const FixupBranch split_access = J_CC(CC_A, Jump::Near);
|
||||||
|
|
||||||
|
// RDX becomes a compact byte index into Cache::data:
|
||||||
|
// set * (8 ways * 32 bytes) + way * 32 + byte offset.
|
||||||
|
MOV(32, R(RSCRATCH), R(RSCRATCH2));
|
||||||
|
AND(32, R(RSCRATCH), Imm32(0xFE0));
|
||||||
|
SHL(32, R(RSCRATCH), Imm8(3));
|
||||||
|
AND(32, R(RSCRATCH2), Imm32(31));
|
||||||
|
OR(32, R(RSCRATCH2), R(RSCRATCH));
|
||||||
|
SHL(32, R(RSCRATCH_EXTRA), Imm8(5));
|
||||||
|
OR(32, R(RSCRATCH2), R(RSCRATCH_EXTRA));
|
||||||
|
|
||||||
|
// Update the exact 8-way pseudo-LRU state. Preserve the compact data index across the lookup.
|
||||||
|
PUSH(RSCRATCH2);
|
||||||
|
MOV(32, R(RSCRATCH), R(RSCRATCH2));
|
||||||
|
SHR(32, R(RSCRATCH), Imm8(8));
|
||||||
|
AND(32, R(RSCRATCH), Imm32(PowerPC::CACHE_SETS - 1));
|
||||||
|
MOV(32, R(RSCRATCH_EXTRA), R(RSCRATCH2));
|
||||||
|
SHR(32, R(RSCRATCH_EXTRA), Imm8(5));
|
||||||
|
AND(32, R(RSCRATCH_EXTRA), Imm32(PowerPC::CACHE_WAYS - 1));
|
||||||
|
MOV(64, R(RSCRATCH2), ImmPtr(cache.plru.data()));
|
||||||
|
MOVZX(32, 8, RSCRATCH2, MComplex(RSCRATCH2, RSCRATCH, SCALE_1, 0));
|
||||||
|
SHL(32, R(RSCRATCH2), Imm8(3));
|
||||||
|
OR(32, R(RSCRATCH2), R(RSCRATCH_EXTRA));
|
||||||
|
MOV(64, R(RSCRATCH_EXTRA), ImmPtr(s_dcache_plru_update.data()));
|
||||||
|
MOVZX(32, 8, RSCRATCH2, MComplex(RSCRATCH_EXTRA, RSCRATCH2, SCALE_1, 0));
|
||||||
|
MOV(64, R(RSCRATCH_EXTRA), ImmPtr(cache.plru.data()));
|
||||||
|
MOV(8, MComplex(RSCRATCH_EXTRA, RSCRATCH, SCALE_1, 0), R(RSCRATCH2));
|
||||||
|
POP(RSCRATCH2);
|
||||||
|
|
||||||
|
if (write)
|
||||||
|
{
|
||||||
|
// Restore, byte-swap and store the guest value, then mark the line dirty.
|
||||||
|
POP(RSCRATCH);
|
||||||
|
BSWAP(32, RSCRATCH);
|
||||||
|
MOV(64, R(RSCRATCH_EXTRA), ImmPtr(cache.data.data()));
|
||||||
|
MOV(32, MComplex(RSCRATCH_EXTRA, RSCRATCH2, SCALE_1, 0), R(RSCRATCH));
|
||||||
|
|
||||||
|
MOV(32, R(RSCRATCH), R(RSCRATCH2));
|
||||||
|
SHR(32, R(RSCRATCH), Imm8(8));
|
||||||
|
AND(32, R(RSCRATCH), Imm32(PowerPC::CACHE_SETS - 1));
|
||||||
|
MOV(32, R(RSCRATCH_EXTRA), R(RSCRATCH2));
|
||||||
|
SHR(32, R(RSCRATCH_EXTRA), Imm8(5));
|
||||||
|
AND(32, R(RSCRATCH_EXTRA), Imm32(PowerPC::CACHE_WAYS - 1));
|
||||||
|
MOV(32, R(RSCRATCH2), Imm32(1));
|
||||||
|
SHL(32, R(RSCRATCH2), R(RSCRATCH_EXTRA));
|
||||||
|
MOV(64, R(RSCRATCH_EXTRA), ImmPtr(cache.modified.data()));
|
||||||
|
OR(8, MComplex(RSCRATCH_EXTRA, RSCRATCH, SCALE_1, 0), R(RSCRATCH2));
|
||||||
|
MOV(32, R(RSCRATCH), Imm32(1));
|
||||||
|
RET();
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
MOV(64, R(RSCRATCH_EXTRA), ImmPtr(cache.data.data()));
|
||||||
|
MOV(32, R(RSCRATCH), MComplex(RSCRATCH_EXTRA, RSCRATCH2, SCALE_1, 0));
|
||||||
|
BSWAP(32, RSCRATCH);
|
||||||
|
ADD(64, R(RSCRATCH), Imm8(1));
|
||||||
|
RET();
|
||||||
|
}
|
||||||
|
|
||||||
|
SetJumpTarget(invalid_bat);
|
||||||
|
SetJumpTarget(not_exram);
|
||||||
|
SetJumpTarget(exram_out_of_range);
|
||||||
|
SetJumpTarget(cache_miss);
|
||||||
|
SetJumpTarget(split_access);
|
||||||
|
if (write)
|
||||||
|
POP(RSCRATCH2);
|
||||||
|
XOR(32, R(RSCRATCH), R(RSCRATCH));
|
||||||
|
RET();
|
||||||
|
|
||||||
|
if (write)
|
||||||
|
Common::JitRegister::Register(start, GetCodePtr(), "JIT_dcache32_write_hit");
|
||||||
|
else
|
||||||
|
Common::JitRegister::Register(start, GetCodePtr(), "JIT_dcache32_read_hit");
|
||||||
|
}
|
||||||
|
|
||||||
// Since the following float conversion functions are used in non-arithmetic PPC float
|
// Since the following float conversion functions are used in non-arithmetic PPC float
|
||||||
// instructions, they must convert floats bitexact and never flush denormals to zero or turn SNaNs
|
// instructions, they must convert floats bitexact and never flush denormals to zero or turn SNaNs
|
||||||
// into QNaNs. This means we can't use CVTSS2SD/CVTSD2SS.
|
// into QNaNs. This means we can't use CVTSS2SD/CVTSD2SS.
|
||||||
|
|||||||
@@ -27,9 +27,16 @@ class CommonAsmRoutines : public CommonAsmRoutinesBase, public QuantizedMemoryRo
|
|||||||
{
|
{
|
||||||
public:
|
public:
|
||||||
explicit CommonAsmRoutines(Jit64& jit) : QuantizedMemoryRoutines(jit) {}
|
explicit CommonAsmRoutines(Jit64& jit) : QuantizedMemoryRoutines(jit) {}
|
||||||
|
|
||||||
|
// Accurate Broadway D-cache leaf routines. Static storage intentionally keeps the layout of
|
||||||
|
// Jit64AsmRoutineManager (and therefore Jit64) unchanged.
|
||||||
|
static const u8* dcache32_read_hit_dbat;
|
||||||
|
static const u8* dcache32_write_hit_dbat;
|
||||||
|
|
||||||
void GenFrsqrte();
|
void GenFrsqrte();
|
||||||
void GenFres();
|
void GenFres();
|
||||||
void GenMfcr();
|
void GenMfcr();
|
||||||
|
void GenDCache32Hit(bool write);
|
||||||
|
|
||||||
protected:
|
protected:
|
||||||
void GenConvertDoubleToSingle();
|
void GenConvertDoubleToSingle();
|
||||||
|
|||||||
@@ -50,6 +50,7 @@
|
|||||||
#include "Core/HW/MMIO.h"
|
#include "Core/HW/MMIO.h"
|
||||||
#include "Core/HW/Memmap.h"
|
#include "Core/HW/Memmap.h"
|
||||||
#include "Core/HW/ProcessorInterface.h"
|
#include "Core/HW/ProcessorInterface.h"
|
||||||
|
#include "Core/HW/VideoInterface.h"
|
||||||
#include "Core/HW/WII_IPC.h"
|
#include "Core/HW/WII_IPC.h"
|
||||||
#include "Core/IOS/Starlet/Starlet.h"
|
#include "Core/IOS/Starlet/Starlet.h"
|
||||||
#include "Core/IOS/Starlet/StarletMemory.h"
|
#include "Core/IOS/Starlet/StarletMemory.h"
|
||||||
@@ -250,6 +251,17 @@ T MMU::ReadFromHardware(u32 em_address)
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// The Wii Menu polls VI_VERTICAL_BEAM_POSITION in a very tight loop while synchronizing its
|
||||||
|
// startup screens. Keep the exact live VI value, but avoid routing every 16-bit read through the
|
||||||
|
// generic MMIO mapping, type-erased handler and std::function layers. Address translation and
|
||||||
|
// all timing updates still happen normally before this point.
|
||||||
|
if constexpr (flag == XCheckTLBFlag::Read && sizeof(T) == sizeof(u16))
|
||||||
|
{
|
||||||
|
constexpr u32 vi_vertical_beam_position = 0x0c002000 | VideoInterface::VI_VERTICAL_BEAM_POSITION;
|
||||||
|
if (em_address == vi_vertical_beam_position)
|
||||||
|
return static_cast<T>(m_system.GetVideoInterface().GetVerticalBeamPosition());
|
||||||
|
}
|
||||||
|
|
||||||
if (flag == XCheckTLBFlag::Read && (em_address & 0xF8000000) == 0x08000000)
|
if (flag == XCheckTLBFlag::Read && (em_address & 0xF8000000) == 0x08000000)
|
||||||
{
|
{
|
||||||
if (em_address < 0x0c000000)
|
if (em_address < 0x0c000000)
|
||||||
@@ -821,6 +833,213 @@ void MMU::Write<u64>(const u64 var, const u32 address)
|
|||||||
WriteToHardware<XCheckTLBFlag::Write>(address + sizeof(u32), static_cast<u32>(var), 4);
|
WriteToHardware<XCheckTLBFlag::Write>(address + sizeof(u32), static_cast<u32>(var), 4);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
template <std::unsigned_integral T>
|
||||||
|
T MMU::ReadForJit(const u32 address)
|
||||||
|
{
|
||||||
|
if (!m_ppc_state.m_enable_dcache || m_power_pc.GetMemChecks().HasAny() ||
|
||||||
|
(address & HW_PAGE_MASK) > HW_PAGE_SIZE - sizeof(T))
|
||||||
|
{
|
||||||
|
return Read<T>(address);
|
||||||
|
}
|
||||||
|
|
||||||
|
u32 physical_address = address;
|
||||||
|
if (m_ppc_state.msr.DR)
|
||||||
|
{
|
||||||
|
const u32 bat_result = m_dbat_table[address >> BAT_INDEX_SHIFT];
|
||||||
|
if ((bat_result & BAT_MAPPED_BIT) == 0 || (bat_result & BAT_WI_BIT) != 0)
|
||||||
|
return Read<T>(address);
|
||||||
|
|
||||||
|
physical_address =
|
||||||
|
(bat_result & BAT_RESULT_MASK) | (address & (BAT_PAGE_SIZE - 1));
|
||||||
|
}
|
||||||
|
|
||||||
|
if (m_memory.GetRAM() && (physical_address & 0xF8000000) == 0)
|
||||||
|
{
|
||||||
|
physical_address &= m_memory.GetRamMask();
|
||||||
|
const T value = m_ppc_state.dCache.ReadMainMemoryValue<T, false>(
|
||||||
|
m_memory, physical_address, HID0(m_ppc_state).DLOCK);
|
||||||
|
return Common::FromBigEndian(value);
|
||||||
|
}
|
||||||
|
|
||||||
|
if (m_memory.GetEXRAM() && (physical_address >> 28) == 1 &&
|
||||||
|
(physical_address & 0x0FFFFFFF) < m_memory.GetExRamSizeReal())
|
||||||
|
{
|
||||||
|
physical_address = (physical_address & 0x0FFFFFFF) + CACHE_EXRAM_BIT;
|
||||||
|
const T value = m_ppc_state.dCache.ReadMainMemoryValue<T, true>(
|
||||||
|
m_memory, physical_address, HID0(m_ppc_state).DLOCK);
|
||||||
|
return Common::FromBigEndian(value);
|
||||||
|
}
|
||||||
|
|
||||||
|
return Read<T>(address);
|
||||||
|
}
|
||||||
|
template u8 MMU::ReadForJit<u8>(u32 address);
|
||||||
|
template u16 MMU::ReadForJit<u16>(u32 address);
|
||||||
|
template u32 MMU::ReadForJit<u32>(u32 address);
|
||||||
|
template u64 MMU::ReadForJit<u64>(u32 address);
|
||||||
|
|
||||||
|
template <std::unsigned_integral T>
|
||||||
|
void MMU::WriteForJit(const Common::MakeAtLeastU32<T> var, const u32 address)
|
||||||
|
{
|
||||||
|
// A 64-bit Broadway store is represented by two 32-bit bus writes in the generic path. Keep
|
||||||
|
// that uncommon operation there; byte, halfword and word animation/state stores use this path.
|
||||||
|
if constexpr (sizeof(T) > sizeof(u32))
|
||||||
|
{
|
||||||
|
Write<T>(var, address);
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
if (!m_ppc_state.m_enable_dcache || m_power_pc.GetMemChecks().HasAny() ||
|
||||||
|
(address & HW_PAGE_MASK) > HW_PAGE_SIZE - sizeof(T))
|
||||||
|
{
|
||||||
|
Write<T>(var, address);
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
u32 physical_address = address;
|
||||||
|
if (m_ppc_state.msr.DR)
|
||||||
|
{
|
||||||
|
const u32 bat_result = m_dbat_table[address >> BAT_INDEX_SHIFT];
|
||||||
|
if ((bat_result & BAT_MAPPED_BIT) == 0 || (bat_result & BAT_WI_BIT) != 0)
|
||||||
|
{
|
||||||
|
Write<T>(var, address);
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
physical_address =
|
||||||
|
(bat_result & BAT_RESULT_MASK) | (address & (BAT_PAGE_SIZE - 1));
|
||||||
|
}
|
||||||
|
|
||||||
|
const T swapped_value = Common::FromBigEndian(static_cast<T>(var));
|
||||||
|
if (m_memory.GetRAM() && (physical_address & 0xF8000000) == 0)
|
||||||
|
{
|
||||||
|
physical_address &= m_memory.GetRamMask();
|
||||||
|
m_ppc_state.dCache.WriteMainMemoryValue<T, false>(
|
||||||
|
m_memory, physical_address, swapped_value, HID0(m_ppc_state).DLOCK);
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
if (m_memory.GetEXRAM() && (physical_address >> 28) == 1 &&
|
||||||
|
(physical_address & 0x0FFFFFFF) < m_memory.GetExRamSizeReal())
|
||||||
|
{
|
||||||
|
physical_address = (physical_address & 0x0FFFFFFF) + CACHE_EXRAM_BIT;
|
||||||
|
m_ppc_state.dCache.WriteMainMemoryValue<T, true>(
|
||||||
|
m_memory, physical_address, swapped_value, HID0(m_ppc_state).DLOCK);
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
Write<T>(var, address);
|
||||||
|
}
|
||||||
|
template void MMU::WriteForJit<u8>(u32 var, u32 address);
|
||||||
|
template void MMU::WriteForJit<u16>(u32 var, u32 address);
|
||||||
|
template void MMU::WriteForJit<u32>(u32 var, u32 address);
|
||||||
|
template void MMU::WriteForJit<u64>(u64 var, u32 address);
|
||||||
|
|
||||||
|
u64 MMU::TryReadDCache32ForJit(const u32 address)
|
||||||
|
{
|
||||||
|
// The generic helper handles the rare split transaction precisely.
|
||||||
|
if ((address & HW_PAGE_MASK) > HW_PAGE_SIZE - sizeof(u32))
|
||||||
|
return 0;
|
||||||
|
|
||||||
|
u32 physical_address = address;
|
||||||
|
if (m_ppc_state.msr.DR)
|
||||||
|
{
|
||||||
|
const u32 bat_result = m_dbat_table[address >> BAT_INDEX_SHIFT];
|
||||||
|
if ((bat_result & (BAT_MAPPED_BIT | BAT_WI_BIT)) != BAT_MAPPED_BIT)
|
||||||
|
return 0;
|
||||||
|
physical_address =
|
||||||
|
(bat_result & BAT_RESULT_MASK) | (address & (BAT_PAGE_SIZE - 1));
|
||||||
|
}
|
||||||
|
|
||||||
|
bool exram = false;
|
||||||
|
u32 lookup_index;
|
||||||
|
if (m_memory.GetRAM() && (physical_address & 0xF8000000) == 0)
|
||||||
|
{
|
||||||
|
physical_address &= m_memory.GetRamMask();
|
||||||
|
lookup_index = physical_address >> 5;
|
||||||
|
}
|
||||||
|
else if (m_memory.GetEXRAM() && (physical_address >> 28) == 1 &&
|
||||||
|
(physical_address & 0x0FFFFFFF) < m_memory.GetExRamSizeReal())
|
||||||
|
{
|
||||||
|
exram = true;
|
||||||
|
physical_address = (physical_address & 0x0FFFFFFF) + CACHE_EXRAM_BIT;
|
||||||
|
lookup_index = (physical_address & 0x0FFFFFFF) >> 5;
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
if ((physical_address & 31) > 32 - sizeof(u32))
|
||||||
|
return 0;
|
||||||
|
|
||||||
|
Cache& cache = m_ppc_state.dCache;
|
||||||
|
const u32 line_address = physical_address & ~31U;
|
||||||
|
const u32 set = (line_address >> 5) & 0x7f;
|
||||||
|
const u32 way = exram ? cache.lookup_table_ex[lookup_index] : cache.lookup_table[lookup_index];
|
||||||
|
if (way == 0xff)
|
||||||
|
return 0;
|
||||||
|
|
||||||
|
cache.plru[set] = (cache.plru[set] & ~Cache::PLRU_MASK[way]) | Cache::PLRU_VALUE[way];
|
||||||
|
u32 value;
|
||||||
|
std::memcpy(&value,
|
||||||
|
reinterpret_cast<const u8*>(cache.data[set][way].data()) +
|
||||||
|
(physical_address & 31),
|
||||||
|
sizeof(value));
|
||||||
|
return static_cast<u64>(Common::swap32(value)) + 1;
|
||||||
|
}
|
||||||
|
|
||||||
|
bool MMU::TryWriteDCache32ForJit(const u32 var, const u32 address)
|
||||||
|
{
|
||||||
|
if ((address & HW_PAGE_MASK) > HW_PAGE_SIZE - sizeof(u32))
|
||||||
|
return false;
|
||||||
|
|
||||||
|
u32 physical_address = address;
|
||||||
|
if (m_ppc_state.msr.DR)
|
||||||
|
{
|
||||||
|
const u32 bat_result = m_dbat_table[address >> BAT_INDEX_SHIFT];
|
||||||
|
if ((bat_result & (BAT_MAPPED_BIT | BAT_WI_BIT)) != BAT_MAPPED_BIT)
|
||||||
|
return false;
|
||||||
|
physical_address =
|
||||||
|
(bat_result & BAT_RESULT_MASK) | (address & (BAT_PAGE_SIZE - 1));
|
||||||
|
}
|
||||||
|
|
||||||
|
bool exram = false;
|
||||||
|
u32 lookup_index;
|
||||||
|
if (m_memory.GetRAM() && (physical_address & 0xF8000000) == 0)
|
||||||
|
{
|
||||||
|
physical_address &= m_memory.GetRamMask();
|
||||||
|
lookup_index = physical_address >> 5;
|
||||||
|
}
|
||||||
|
else if (m_memory.GetEXRAM() && (physical_address >> 28) == 1 &&
|
||||||
|
(physical_address & 0x0FFFFFFF) < m_memory.GetExRamSizeReal())
|
||||||
|
{
|
||||||
|
exram = true;
|
||||||
|
physical_address = (physical_address & 0x0FFFFFFF) + CACHE_EXRAM_BIT;
|
||||||
|
lookup_index = (physical_address & 0x0FFFFFFF) >> 5;
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
if ((physical_address & 31) > 32 - sizeof(u32))
|
||||||
|
return false;
|
||||||
|
|
||||||
|
Cache& cache = m_ppc_state.dCache;
|
||||||
|
const u32 line_address = physical_address & ~31U;
|
||||||
|
const u32 set = (line_address >> 5) & 0x7f;
|
||||||
|
const u32 way = exram ? cache.lookup_table_ex[lookup_index] : cache.lookup_table[lookup_index];
|
||||||
|
if (way == 0xff)
|
||||||
|
return false;
|
||||||
|
|
||||||
|
cache.plru[set] = (cache.plru[set] & ~Cache::PLRU_MASK[way]) | Cache::PLRU_VALUE[way];
|
||||||
|
const u32 swapped_value = Common::swap32(var);
|
||||||
|
std::memcpy(reinterpret_cast<u8*>(cache.data[set][way].data()) + (physical_address & 31),
|
||||||
|
&swapped_value, sizeof(swapped_value));
|
||||||
|
cache.modified[set] |= 1U << way;
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
void MMU::Write_U16_Swap(const u32 var, const u32 address)
|
void MMU::Write_U16_Swap(const u32 var, const u32 address)
|
||||||
{
|
{
|
||||||
Write<u16>((var & 0xFFFF0000) | Common::swap16(static_cast<u16>(var)), address);
|
Write<u16>((var & 0xFFFF0000) | Common::swap16(static_cast<u16>(var)), address);
|
||||||
@@ -2183,10 +2402,18 @@ void ClearDCacheLineFromJit(MMU& mmu, u32 address)
|
|||||||
{
|
{
|
||||||
mmu.ClearDCacheLine(address);
|
mmu.ClearDCacheLine(address);
|
||||||
}
|
}
|
||||||
|
u64 TryReadDCache32FromJit(MMU& mmu, u32 address)
|
||||||
|
{
|
||||||
|
return mmu.TryReadDCache32ForJit(address);
|
||||||
|
}
|
||||||
|
bool TryWriteDCache32FromJit(MMU& mmu, u32 var, u32 address)
|
||||||
|
{
|
||||||
|
return mmu.TryWriteDCache32ForJit(var, address);
|
||||||
|
}
|
||||||
template <std::unsigned_integral T>
|
template <std::unsigned_integral T>
|
||||||
Common::MakeAtLeastU32<T> ReadFromJit(MMU& mmu, u32 address)
|
Common::MakeAtLeastU32<T> ReadFromJit(MMU& mmu, u32 address)
|
||||||
{
|
{
|
||||||
return mmu.Read<T>(address);
|
return mmu.ReadForJit<T>(address);
|
||||||
}
|
}
|
||||||
template u32 ReadFromJit<u8>(MMU& mmu, u32 address);
|
template u32 ReadFromJit<u8>(MMU& mmu, u32 address);
|
||||||
template u32 ReadFromJit<u16>(MMU& mmu, u32 address);
|
template u32 ReadFromJit<u16>(MMU& mmu, u32 address);
|
||||||
@@ -2196,7 +2423,7 @@ template u64 ReadFromJit<u64>(MMU& mmu, u32 address);
|
|||||||
template <std::unsigned_integral T>
|
template <std::unsigned_integral T>
|
||||||
void WriteFromJit(MMU& mmu, Common::MakeAtLeastU32<T> var, u32 address)
|
void WriteFromJit(MMU& mmu, Common::MakeAtLeastU32<T> var, u32 address)
|
||||||
{
|
{
|
||||||
mmu.Write<T>(var, address);
|
mmu.WriteForJit<T>(var, address);
|
||||||
}
|
}
|
||||||
template void WriteFromJit<u8>(MMU& mmu, u32 var, u32 address);
|
template void WriteFromJit<u8>(MMU& mmu, u32 var, u32 address);
|
||||||
template void WriteFromJit<u16>(MMU& mmu, u32 var, u32 address);
|
template void WriteFromJit<u16>(MMU& mmu, u32 var, u32 address);
|
||||||
|
|||||||
@@ -233,6 +233,21 @@ public:
|
|||||||
template <std::unsigned_integral T>
|
template <std::unsigned_integral T>
|
||||||
void Write(Common::MakeAtLeastU32<T> var, u32 address);
|
void Write(Common::MakeAtLeastU32<T> var, u32 address);
|
||||||
|
|
||||||
|
// JIT memory helpers. When accurate D-cache emulation is enabled, ordinary fastmem cannot be
|
||||||
|
// used because Broadway loads and stores must update the emulated cache rather than backing RAM.
|
||||||
|
// These helpers retain those exact semantics while bypassing the generic hardware dispatcher for
|
||||||
|
// the overwhelmingly common BAT-mapped MEM1/MEM2 case.
|
||||||
|
template <std::unsigned_integral T>
|
||||||
|
T ReadForJit(u32 address);
|
||||||
|
template <std::unsigned_integral T>
|
||||||
|
void WriteForJit(Common::MakeAtLeastU32<T> var, u32 address);
|
||||||
|
|
||||||
|
// Leaf fast paths used by Jit64 when accurate D-cache emulation is active. A zero/false result
|
||||||
|
// means that the caller must use ReadForJit/WriteForJit (cache miss, MMIO, TLB, WI, etc.). Reads
|
||||||
|
// encode a successful 32-bit value as value+1 in 64 bits so every guest value remains representable.
|
||||||
|
u64 TryReadDCache32ForJit(u32 address);
|
||||||
|
bool TryWriteDCache32ForJit(u32 var, u32 address);
|
||||||
|
|
||||||
void Write_U16_Swap(u32 var, u32 address);
|
void Write_U16_Swap(u32 var, u32 address);
|
||||||
void Write_U32_Swap(u32 var, u32 address);
|
void Write_U32_Swap(u32 var, u32 address);
|
||||||
void Write_U64_Swap(u64 var, u32 address);
|
void Write_U64_Swap(u64 var, u32 address);
|
||||||
@@ -374,6 +389,8 @@ private:
|
|||||||
};
|
};
|
||||||
|
|
||||||
void ClearDCacheLineFromJit(MMU& mmu, u32 address);
|
void ClearDCacheLineFromJit(MMU& mmu, u32 address);
|
||||||
|
u64 TryReadDCache32FromJit(MMU& mmu, u32 address);
|
||||||
|
bool TryWriteDCache32FromJit(MMU& mmu, u32 var, u32 address);
|
||||||
template <std::unsigned_integral T>
|
template <std::unsigned_integral T>
|
||||||
// Returns zero-extended value
|
// Returns zero-extended value
|
||||||
Common::MakeAtLeastU32<T> ReadFromJit(MMU& mmu, u32 address);
|
Common::MakeAtLeastU32<T> ReadFromJit(MMU& mmu, u32 address);
|
||||||
|
|||||||
@@ -734,10 +734,11 @@ void PPCAnalyzer::SetInstructionStats(CodeBlock* block, CodeOp* code,
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
bool PPCAnalyzer::IsBusyWaitLoop(CodeBlock* block, CodeOp* code, size_t instructions) const
|
bool PPCAnalyzer::IsBusyWaitLoop(CodeOp* code, size_t loop_start, size_t branch_index) const
|
||||||
{
|
{
|
||||||
// Very basic algorithm to detect busy wait loops:
|
// Very basic algorithm to detect busy wait loops:
|
||||||
// * It loops to itself and does not contain any other branches.
|
// * It loops to the first instruction in the candidate range and does not contain any other
|
||||||
|
// branches.
|
||||||
// * It does not write to memory.
|
// * It does not write to memory.
|
||||||
// * It only reads from registers it wrote to earlier in the loop, or it
|
// * It only reads from registers it wrote to earlier in the loop, or it
|
||||||
// does not write to these registers.
|
// does not write to these registers.
|
||||||
@@ -748,14 +749,15 @@ bool PPCAnalyzer::IsBusyWaitLoop(CodeBlock* block, CodeOp* code, size_t instruct
|
|||||||
// don't detect these at the moment.
|
// don't detect these at the moment.
|
||||||
std::bitset<32> write_disallowed_regs;
|
std::bitset<32> write_disallowed_regs;
|
||||||
std::bitset<32> written_regs;
|
std::bitset<32> written_regs;
|
||||||
for (size_t i = 0; i <= instructions; ++i)
|
for (size_t i = loop_start; i <= branch_index; ++i)
|
||||||
{
|
{
|
||||||
if (code[i].opinfo->type == OpType::Branch)
|
if (code[i].opinfo->type == OpType::Branch)
|
||||||
{
|
{
|
||||||
if (code[i].branchUsesCtr)
|
if (code[i].branchUsesCtr)
|
||||||
return false;
|
return false;
|
||||||
if (code[i].branchTo == block->m_address && i == instructions)
|
if (code[i].branchTo == code[loop_start].address && i == branch_index)
|
||||||
return true;
|
return true;
|
||||||
|
return false;
|
||||||
}
|
}
|
||||||
// A `nop` is actually a `ori r0, r0, 0`, which would violate the rules (unless `r0` was written
|
// A `nop` is actually a `ori r0, r0, 0`, which would violate the rules (unless `r0` was written
|
||||||
// earlier).
|
// earlier).
|
||||||
@@ -763,6 +765,12 @@ bool PPCAnalyzer::IsBusyWaitLoop(CodeBlock* block, CodeOp* code, size_t instruct
|
|||||||
{
|
{
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
|
// The JIT implements sync as a host no-op. It is commonly placed directly after an MMIO load
|
||||||
|
// in hardware polling loops, and does not make a load-only loop unsafe to idle.
|
||||||
|
else if (code[i].inst.OPCD == 31 && code[i].inst.SUBOP10 == 598)
|
||||||
|
{
|
||||||
|
continue;
|
||||||
|
}
|
||||||
else if (code[i].opinfo->type != OpType::Integer && code[i].opinfo->type != OpType::Load)
|
else if (code[i].opinfo->type != OpType::Integer && code[i].opinfo->type != OpType::Load)
|
||||||
{
|
{
|
||||||
// In the future, some subsets of other instruction types might get
|
// In the future, some subsets of other instruction types might get
|
||||||
@@ -946,8 +954,22 @@ u32 PPCAnalyzer::Analyze(u32 address, CodeBlock* block, CodeBuffer* buffer,
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
code[i].branchIsIdleLoop =
|
code[i].branchIsIdleLoop = false;
|
||||||
code[i].branchTo == block->m_address && IsBusyWaitLoop(block, code, i);
|
if (code[i].opinfo->type == OpType::Branch)
|
||||||
|
{
|
||||||
|
// Branch following can inline a small polling function into its caller. Such a loop branches
|
||||||
|
// to an instruction in the middle of the analyzed block rather than block->m_address, so the
|
||||||
|
// old detector missed it. Locate the innermost matching instruction and validate only that
|
||||||
|
// loop range.
|
||||||
|
for (size_t candidate = i + 1; candidate-- > 0;)
|
||||||
|
{
|
||||||
|
if (code[candidate].address == code[i].branchTo)
|
||||||
|
{
|
||||||
|
code[i].branchIsIdleLoop = IsBusyWaitLoop(code, candidate, i);
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
if (follow && numFollows < BRANCH_FOLLOWING_THRESHOLD)
|
if (follow && numFollows < BRANCH_FOLLOWING_THRESHOLD)
|
||||||
{
|
{
|
||||||
|
|||||||
@@ -199,7 +199,7 @@ private:
|
|||||||
ReorderType type) const;
|
ReorderType type) const;
|
||||||
void ReorderInstructions(u32 instructions, CodeOp* code) const;
|
void ReorderInstructions(u32 instructions, CodeOp* code) const;
|
||||||
void SetInstructionStats(CodeBlock* block, CodeOp* code, const GekkoOPInfo* opinfo) const;
|
void SetInstructionStats(CodeBlock* block, CodeOp* code, const GekkoOPInfo* opinfo) const;
|
||||||
bool IsBusyWaitLoop(CodeBlock* block, CodeOp* code, size_t instructions) const;
|
bool IsBusyWaitLoop(CodeOp* code, size_t loop_start, size_t branch_index) const;
|
||||||
|
|
||||||
// Options
|
// Options
|
||||||
u32 m_options = 0;
|
u32 m_options = 0;
|
||||||
|
|||||||
@@ -275,10 +275,14 @@ void PowerPCManager::Init(CPUCore cpu_core)
|
|||||||
|
|
||||||
Reset();
|
Reset();
|
||||||
|
|
||||||
InitializeCPUCore(cpu_core);
|
|
||||||
auto& memory = m_system.GetMemory();
|
auto& memory = m_system.GetMemory();
|
||||||
m_ppc_state.iCache.Init(memory);
|
m_ppc_state.iCache.Init(memory);
|
||||||
m_ppc_state.dCache.Init(memory);
|
m_ppc_state.dCache.Init(memory);
|
||||||
|
|
||||||
|
// JIT common routines may embed stable pointers into the cache lookup tables. Allocate those
|
||||||
|
// tables before generating a CPU core, rather than leaving the JIT with the empty vectors from
|
||||||
|
// PowerPCState construction. Reset and state loads only refill these allocations afterwards.
|
||||||
|
InitializeCPUCore(cpu_core);
|
||||||
}
|
}
|
||||||
|
|
||||||
void PowerPCManager::Reset()
|
void PowerPCManager::Reset()
|
||||||
|
|||||||
@@ -97,7 +97,7 @@ struct CompressAndDumpStateArgs
|
|||||||
static Common::WorkQueueThreadSP<CompressAndDumpStateArgs> s_compress_and_dump_thread;
|
static Common::WorkQueueThreadSP<CompressAndDumpStateArgs> s_compress_and_dump_thread;
|
||||||
|
|
||||||
// Don't forget to increase this after doing changes on the savestate system
|
// Don't forget to increase this after doing changes on the savestate system
|
||||||
constexpr u32 STATE_VERSION = 192; // Last changed in PR 14646
|
constexpr u32 STATE_VERSION = 193; // Starlet HW_TIMER counter offset.
|
||||||
|
|
||||||
// Increase this if the StateExtendedHeader definition changes
|
// Increase this if the StateExtendedHeader definition changes
|
||||||
constexpr u32 EXTENDED_HEADER_VERSION = 1; // Last changed in PR 12217
|
constexpr u32 EXTENDED_HEADER_VERSION = 1; // Last changed in PR 12217
|
||||||
|
|||||||
@@ -26,6 +26,7 @@ if(_M_X86_64)
|
|||||||
PowerPC/DivUtilsTest.cpp
|
PowerPC/DivUtilsTest.cpp
|
||||||
PowerPC/PageTableHostMappingTest.cpp
|
PowerPC/PageTableHostMappingTest.cpp
|
||||||
PowerPC/Jit64Common/ConvertDoubleToSingle.cpp
|
PowerPC/Jit64Common/ConvertDoubleToSingle.cpp
|
||||||
|
PowerPC/Jit64Common/DCache.cpp
|
||||||
PowerPC/Jit64Common/Fres.cpp
|
PowerPC/Jit64Common/Fres.cpp
|
||||||
PowerPC/Jit64Common/Frsqrte.cpp
|
PowerPC/Jit64Common/Frsqrte.cpp
|
||||||
)
|
)
|
||||||
|
|||||||
@@ -10,6 +10,7 @@
|
|||||||
|
|
||||||
#include <gtest/gtest.h>
|
#include <gtest/gtest.h>
|
||||||
|
|
||||||
|
#include "Common/ChunkFile.h"
|
||||||
#include "Common/CommonTypes.h"
|
#include "Common/CommonTypes.h"
|
||||||
#include "Core/Core.h"
|
#include "Core/Core.h"
|
||||||
#include "Core/HW/WII_IPC.h"
|
#include "Core/HW/WII_IPC.h"
|
||||||
@@ -171,6 +172,14 @@ public:
|
|||||||
return m_sram_fastmem_enabled ? &m_sram_split_mode : nullptr;
|
return m_sram_fastmem_enabled ? &m_sram_split_mode : nullptr;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
const u8* GetDirectMemoryPointer(u32 address, u32 size) const override
|
||||||
|
{
|
||||||
|
const size_t offset = ToOffset(address);
|
||||||
|
if (size == 0 || offset > m_memory.size() || size > m_memory.size() - offset)
|
||||||
|
return nullptr;
|
||||||
|
return m_memory.data() + offset;
|
||||||
|
}
|
||||||
|
|
||||||
void SetIdlePollSafe(bool safe) { m_idle_poll_safe = safe; }
|
void SetIdlePollSafe(bool safe) { m_idle_poll_safe = safe; }
|
||||||
void SetSliceStablePollAddress(u32 address)
|
void SetSliceStablePollAddress(u32 address)
|
||||||
{
|
{
|
||||||
@@ -376,7 +385,7 @@ TEST(StarletTimer, ZeroDelayAlarmMatchesImmediatelyAndUsesIRQW1C)
|
|||||||
memory.Write8(address + 3, static_cast<u8>(value));
|
memory.Write8(address + 3, static_cast<u8>(value));
|
||||||
};
|
};
|
||||||
|
|
||||||
memory.AdvanceCycles(405);
|
memory.AdvanceCycles(32 * 128);
|
||||||
ASSERT_EQ(read_word(timer), 32u);
|
ASSERT_EQ(read_word(timer), 32u);
|
||||||
write_word(alarm, read_word(timer));
|
write_word(alarm, read_word(timer));
|
||||||
EXPECT_EQ(system.GetWiiIPC().ReadStarletRegister(0x38) & INT_CAUSE_TIMER, INT_CAUSE_TIMER);
|
EXPECT_EQ(system.GetWiiIPC().ReadStarletRegister(0x38) & INT_CAUSE_TIMER, INT_CAUSE_TIMER);
|
||||||
@@ -388,12 +397,155 @@ TEST(StarletTimer, ZeroDelayAlarmMatchesImmediatelyAndUsesIRQW1C)
|
|||||||
write_word(arm_irq_flag, INT_CAUSE_TIMER);
|
write_word(arm_irq_flag, INT_CAUSE_TIMER);
|
||||||
EXPECT_EQ(system.GetWiiIPC().ReadStarletRegister(0x38) & INT_CAUSE_TIMER, 0u);
|
EXPECT_EQ(system.GetWiiIPC().ReadStarletRegister(0x38) & INT_CAUSE_TIMER, 0u);
|
||||||
|
|
||||||
memory.AdvanceCycles(404);
|
memory.AdvanceCycles(32 * 128 - 1);
|
||||||
EXPECT_EQ(system.GetWiiIPC().ReadStarletRegister(0x38) & INT_CAUSE_TIMER, 0u);
|
EXPECT_EQ(system.GetWiiIPC().ReadStarletRegister(0x38) & INT_CAUSE_TIMER, 0u);
|
||||||
memory.AdvanceCycles(1);
|
memory.AdvanceCycles(1);
|
||||||
EXPECT_EQ(system.GetWiiIPC().ReadStarletRegister(0x38) & INT_CAUSE_TIMER, INT_CAUSE_TIMER);
|
EXPECT_EQ(system.GetWiiIPC().ReadStarletRegister(0x38) & INT_CAUSE_TIMER, INT_CAUSE_TIMER);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
TEST(StarletRegisters, WideTimerAndInterruptAccessesMatchHardwareSemantics)
|
||||||
|
{
|
||||||
|
constexpr u32 hardware_base = 0x0d800000;
|
||||||
|
constexpr u32 timer = hardware_base + 0x10;
|
||||||
|
constexpr u32 arm_irq_flag = hardware_base + 0x38;
|
||||||
|
constexpr u32 arm_irq_mask = hardware_base + 0x3c;
|
||||||
|
Core::DeclareAsCPUThread();
|
||||||
|
auto& system = Core::System::GetInstance();
|
||||||
|
system.GetWiiIPC().Reset();
|
||||||
|
StarletMemory memory(system);
|
||||||
|
memory.Reset();
|
||||||
|
|
||||||
|
memory.AdvanceCycles(32 * 128);
|
||||||
|
EXPECT_EQ(memory.Read32(timer), 32u);
|
||||||
|
memory.Write32(timer, 64);
|
||||||
|
EXPECT_EQ(memory.Read32(timer), 64u);
|
||||||
|
|
||||||
|
system.GetWiiIPC().SetStarletInterrupt(INT_CAUSE_TIMER, true);
|
||||||
|
EXPECT_EQ(memory.Read32(arm_irq_flag) & INT_CAUSE_TIMER, INT_CAUSE_TIMER);
|
||||||
|
memory.Write32(arm_irq_flag, INT_CAUSE_TIMER);
|
||||||
|
EXPECT_EQ(memory.Read32(arm_irq_flag) & INT_CAUSE_TIMER, 0u);
|
||||||
|
|
||||||
|
memory.Write32(arm_irq_mask, 0x800619ef);
|
||||||
|
EXPECT_EQ(memory.Read32(arm_irq_mask), 0x800619efu);
|
||||||
|
}
|
||||||
|
|
||||||
|
TEST(StarletTimer, RunsAtOneTickPer128ARMCycles)
|
||||||
|
{
|
||||||
|
constexpr u32 timer = 0x0d800010;
|
||||||
|
Core::DeclareAsCPUThread();
|
||||||
|
auto& system = Core::System::GetInstance();
|
||||||
|
system.GetWiiIPC().Reset();
|
||||||
|
StarletMemory memory(system);
|
||||||
|
memory.Reset();
|
||||||
|
|
||||||
|
memory.AdvanceCycles(127);
|
||||||
|
EXPECT_EQ(memory.Read32(timer), 0u);
|
||||||
|
memory.AdvanceCycles(1);
|
||||||
|
EXPECT_EQ(memory.Read32(timer), 1u);
|
||||||
|
memory.AdvanceCycles(243'000'000 - 128);
|
||||||
|
EXPECT_EQ(memory.Read32(timer), 1'898'437u);
|
||||||
|
memory.AdvanceCycles(243'000'000);
|
||||||
|
EXPECT_EQ(memory.Read32(timer), 3'796'875u);
|
||||||
|
}
|
||||||
|
|
||||||
|
TEST(StarletTimer, SchedulerSlicePartitionDoesNotChangeClock)
|
||||||
|
{
|
||||||
|
Core::DeclareAsCPUThread();
|
||||||
|
auto& system = Core::System::GetInstance();
|
||||||
|
StarletMemory memory(system);
|
||||||
|
memory.Reset();
|
||||||
|
|
||||||
|
constexpr u64 total_cycles = 243'000'000;
|
||||||
|
// Active, IPC, and idle scheduler slices must use the same clock, with no
|
||||||
|
// fractional timer ticks lost at the end of a slice.
|
||||||
|
for (const u64 slice : {256u, 4096u, 24300u})
|
||||||
|
{
|
||||||
|
memory.Reset();
|
||||||
|
for (u64 elapsed = 0; elapsed < total_cycles;)
|
||||||
|
{
|
||||||
|
const u64 step = std::min(slice, total_cycles - elapsed);
|
||||||
|
memory.AdvanceCycles(step);
|
||||||
|
elapsed += step;
|
||||||
|
}
|
||||||
|
EXPECT_EQ(memory.Read32(0x0d800010), 1'898'437u) << "slice=" << slice;
|
||||||
|
EXPECT_EQ(memory.GetCycles(), total_cycles);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
TEST(StarletTimer, CounterWritesDoNotRewindPeripheralClock)
|
||||||
|
{
|
||||||
|
constexpr u32 timer = 0x0d800010;
|
||||||
|
Core::DeclareAsCPUThread();
|
||||||
|
auto& system = Core::System::GetInstance();
|
||||||
|
StarletMemory memory(system);
|
||||||
|
memory.Reset();
|
||||||
|
memory.AdvanceCycles(1025);
|
||||||
|
|
||||||
|
for (const u32 value : {1u, 0xffffffffu, 0u, 0x12345678u})
|
||||||
|
{
|
||||||
|
memory.Write32(timer, value);
|
||||||
|
EXPECT_EQ(memory.Read32(timer), value);
|
||||||
|
EXPECT_EQ(memory.GetCycles(), 1025u);
|
||||||
|
}
|
||||||
|
// Reprogramming HW_TIMER leaves the free-running /128 clock phase intact.
|
||||||
|
memory.AdvanceCycles(126);
|
||||||
|
EXPECT_EQ(memory.Read32(timer), 0x12345678u);
|
||||||
|
memory.AdvanceCycles(1);
|
||||||
|
EXPECT_EQ(memory.Read32(timer), 0x12345679u);
|
||||||
|
}
|
||||||
|
|
||||||
|
TEST(StarletTimer, ByteAssembledCounterWritesMatchWideWrites)
|
||||||
|
{
|
||||||
|
constexpr u32 timer = 0x0d800010;
|
||||||
|
Core::DeclareAsCPUThread();
|
||||||
|
auto& system = Core::System::GetInstance();
|
||||||
|
StarletMemory memory(system);
|
||||||
|
memory.Reset();
|
||||||
|
memory.AdvanceCycles(1280);
|
||||||
|
memory.Write8(timer, 0x12);
|
||||||
|
memory.Write8(timer + 1, 0x34);
|
||||||
|
memory.Write8(timer + 2, 0x56);
|
||||||
|
memory.Write8(timer + 3, 0x78);
|
||||||
|
EXPECT_EQ(memory.Read32(timer), 0x12345678u);
|
||||||
|
EXPECT_EQ(memory.GetCycles(), 1280u);
|
||||||
|
memory.AdvanceCycles(128);
|
||||||
|
EXPECT_EQ(memory.Read32(timer), 0x12345679u);
|
||||||
|
}
|
||||||
|
|
||||||
|
TEST(StarletTimer, AlarmFiresAcrossCounterWrap)
|
||||||
|
{
|
||||||
|
constexpr u32 timer = 0x0d800010;
|
||||||
|
constexpr u32 alarm = 0x0d800014;
|
||||||
|
Core::DeclareAsCPUThread();
|
||||||
|
auto& system = Core::System::GetInstance();
|
||||||
|
system.GetWiiIPC().Reset();
|
||||||
|
StarletMemory memory(system);
|
||||||
|
memory.Reset();
|
||||||
|
memory.Write32(timer, 0xfffffffe);
|
||||||
|
memory.Write32(alarm, 1);
|
||||||
|
memory.AdvanceCycles(3 * 128 - 1);
|
||||||
|
EXPECT_EQ(memory.Read32(timer), 0u);
|
||||||
|
EXPECT_EQ(system.GetWiiIPC().ReadStarletRegister(0x38) & INT_CAUSE_TIMER, 0u);
|
||||||
|
memory.AdvanceCycles(1);
|
||||||
|
EXPECT_EQ(memory.Read32(timer), 1u);
|
||||||
|
EXPECT_EQ(system.GetWiiIPC().ReadStarletRegister(0x38) & INT_CAUSE_TIMER, INT_CAUSE_TIMER);
|
||||||
|
}
|
||||||
|
|
||||||
|
TEST(StarletTimer, ResetClearsCounterOffset)
|
||||||
|
{
|
||||||
|
Core::DeclareAsCPUThread();
|
||||||
|
auto& system = Core::System::GetInstance();
|
||||||
|
StarletMemory memory(system);
|
||||||
|
memory.Reset();
|
||||||
|
memory.AdvanceCycles(512);
|
||||||
|
memory.Write32(0x0d800010, 0x12345678);
|
||||||
|
memory.Reset();
|
||||||
|
EXPECT_EQ(memory.GetCycles(), 0u);
|
||||||
|
EXPECT_EQ(memory.Read32(0x0d800010), 0u);
|
||||||
|
memory.AdvanceCycles(128);
|
||||||
|
EXPECT_EQ(memory.Read32(0x0d800010), 1u);
|
||||||
|
}
|
||||||
|
|
||||||
TEST(StarletNAND, HardwareResetPreservesProgrammedFlash)
|
TEST(StarletNAND, HardwareResetPreservesProgrammedFlash)
|
||||||
{
|
{
|
||||||
Core::DeclareAsCPUThread();
|
Core::DeclareAsCPUThread();
|
||||||
@@ -426,6 +578,36 @@ TEST(StarletNAND, HardwareResetPreservesProgrammedFlash)
|
|||||||
EXPECT_EQ(memory.Read32(sram), 0x12345678u);
|
EXPECT_EQ(memory.Read32(sram), 0x12345678u);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
TEST(StarletTimer, StateRoundTripPreservesOffsetAndDividerPhase)
|
||||||
|
{
|
||||||
|
constexpr u32 timer = 0x0d800010;
|
||||||
|
Core::DeclareAsCPUThread();
|
||||||
|
auto& system = Core::System::GetInstance();
|
||||||
|
StarletMemory memory(system);
|
||||||
|
memory.Reset();
|
||||||
|
memory.AdvanceCycles(1025);
|
||||||
|
memory.Write32(timer, 0x12345678);
|
||||||
|
|
||||||
|
std::vector<u8> state_buffer(1024 * 1024);
|
||||||
|
u8* state_pointer = state_buffer.data();
|
||||||
|
PointerWrap writer(&state_pointer, state_buffer.size(), PointerWrap::Mode::Write);
|
||||||
|
memory.DoState(writer);
|
||||||
|
ASSERT_TRUE(writer.IsWriteMode());
|
||||||
|
const size_t state_size = state_pointer - state_buffer.data();
|
||||||
|
|
||||||
|
memory.Reset();
|
||||||
|
state_pointer = state_buffer.data();
|
||||||
|
PointerWrap reader(&state_pointer, state_size, PointerWrap::Mode::Read);
|
||||||
|
memory.DoState(reader);
|
||||||
|
ASSERT_TRUE(reader.IsReadMode());
|
||||||
|
EXPECT_EQ(memory.GetCycles(), 1025u);
|
||||||
|
EXPECT_EQ(memory.Read32(timer), 0x12345678u);
|
||||||
|
memory.AdvanceCycles(126);
|
||||||
|
EXPECT_EQ(memory.Read32(timer), 0x12345678u);
|
||||||
|
memory.AdvanceCycles(1);
|
||||||
|
EXPECT_EQ(memory.Read32(timer), 0x12345679u);
|
||||||
|
}
|
||||||
|
|
||||||
TEST(StarletGPIO, InterruptFlagIsWriteOneToClear)
|
TEST(StarletGPIO, InterruptFlagIsWriteOneToClear)
|
||||||
{
|
{
|
||||||
constexpr u32 hardware_base = 0x0d800000;
|
constexpr u32 hardware_base = 0x0d800000;
|
||||||
@@ -984,6 +1166,35 @@ TEST(StarletARMCore, JitCompilesDrainWriteBufferNatively)
|
|||||||
EXPECT_EQ(core.GetJitFallbackInstructionCount(), 1u);
|
EXPECT_EQ(core.GetJitFallbackInstructionCount(), 1u);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
TEST(StarletARMCore, JitCompilesHotCP15MaintenanceNatively)
|
||||||
|
{
|
||||||
|
#if defined(_M_X86_64)
|
||||||
|
TestBus interpreter_bus;
|
||||||
|
TestBus jit_bus;
|
||||||
|
ARMCore interpreter(interpreter_bus);
|
||||||
|
ARMCore jit(jit_bus);
|
||||||
|
jit.SetJitEnabled(true);
|
||||||
|
const auto install_program = [](TestBus& bus) {
|
||||||
|
bus.WriteARM(0x00, 0xee033f10); // mcr p15, 0, r3, c3, c0, 0 (DACR)
|
||||||
|
bus.WriteARM(0x04, 0xee070f36); // mcr p15, 0, r0, c7, c6, 1
|
||||||
|
bus.WriteARM(0x08, 0xee070f3a); // mcr p15, 0, r0, c7, c10, 1
|
||||||
|
bus.WriteARM(0x0c, 0xeafffffe); // b .
|
||||||
|
};
|
||||||
|
install_program(interpreter_bus);
|
||||||
|
install_program(jit_bus);
|
||||||
|
interpreter.SetRegister(3, 0x55555555);
|
||||||
|
jit.SetRegister(3, 0x55555555);
|
||||||
|
|
||||||
|
ASSERT_EQ(interpreter.RunCycles(4), 4u);
|
||||||
|
ASSERT_EQ(jit.RunCycles(4), 4u);
|
||||||
|
EXPECT_EQ(jit.GetCP15State().domain_access_control,
|
||||||
|
interpreter.GetCP15State().domain_access_control);
|
||||||
|
EXPECT_EQ(jit.GetRegister(15), interpreter.GetRegister(15));
|
||||||
|
EXPECT_EQ(jit.GetJitFallbackInstructionCount(), 0u);
|
||||||
|
EXPECT_EQ(jit.GetJitNativeExecutedInstructions(), 4u);
|
||||||
|
#endif
|
||||||
|
}
|
||||||
|
|
||||||
TEST(StarletARMCore, JitDefersCP15CacheInvalidationUntilTheHostBlockReturns)
|
TEST(StarletARMCore, JitDefersCP15CacheInvalidationUntilTheHostBlockReturns)
|
||||||
{
|
{
|
||||||
TestBus bus;
|
TestBus bus;
|
||||||
@@ -1041,18 +1252,324 @@ TEST(StarletARMCore, JitPreservedBlocksUseCurrentTLBGenerationForFastmem)
|
|||||||
|
|
||||||
EXPECT_EQ(core.RunCycles(2), 2u);
|
EXPECT_EQ(core.RunCycles(2), 2u);
|
||||||
ASSERT_EQ(core.GetRegister(1), 0x11223344u);
|
ASSERT_EQ(core.GetRegister(1), 0x11223344u);
|
||||||
ASSERT_EQ(core.GetJitFallbackInstructionCount(), 1u);
|
ASSERT_EQ(core.GetJitFallbackInstructionCount(), 0u);
|
||||||
ASSERT_EQ(core.GetJitCompiledBlockCount(), 1u);
|
ASSERT_EQ(core.GetJitCompiledBlockCount(), 1u);
|
||||||
|
|
||||||
core.SetRegister(1, 0);
|
core.SetRegister(1, 0);
|
||||||
core.SetRegister(15, 0x80000000);
|
core.SetRegister(15, 0x80000000);
|
||||||
EXPECT_EQ(core.RunCycles(1), 1u);
|
EXPECT_EQ(core.RunCycles(1), 1u);
|
||||||
EXPECT_EQ(core.GetRegister(1), 0x11223344u);
|
EXPECT_EQ(core.GetRegister(1), 0x11223344u);
|
||||||
EXPECT_EQ(core.GetJitFallbackInstructionCount(), 1u);
|
EXPECT_EQ(core.GetJitFallbackInstructionCount(), 0u);
|
||||||
EXPECT_EQ(core.GetJitCompiledBlockCount(), 1u);
|
EXPECT_EQ(core.GetJitCompiledBlockCount(), 1u);
|
||||||
#endif
|
#endif
|
||||||
}
|
}
|
||||||
|
|
||||||
|
TEST(StarletARMCore, JitTLBRevalidationIsSharedByNativeBlocksOnTheSamePage)
|
||||||
|
{
|
||||||
|
#if defined(_M_X86_64)
|
||||||
|
TestBus bus(0x10000);
|
||||||
|
ARMCore core(bus);
|
||||||
|
bus.WriteARM(0x0000, 0xe3a01001); // mov r1, #1
|
||||||
|
bus.WriteARM(0x0020, 0xe3a02002); // mov r2, #2
|
||||||
|
bus.WriteARM(0x0040, 0xee080f17); // invalidate unified TLB
|
||||||
|
bus.WriteARM(0x6000, 0x00000c02); // VA 0x80000000 section -> PA 0
|
||||||
|
core.GetCP15State().translation_table_base = 0x4000;
|
||||||
|
core.GetCP15State().domain_access_control = 3;
|
||||||
|
core.GetCP15State().control |= 1;
|
||||||
|
core.SetJitEnabled(true);
|
||||||
|
|
||||||
|
core.SetRegister(15, 0x80000000);
|
||||||
|
EXPECT_EQ(core.RunCycles(1), 1u);
|
||||||
|
core.SetRegister(15, 0x80000020);
|
||||||
|
EXPECT_EQ(core.RunCycles(1), 1u);
|
||||||
|
core.SetRegister(15, 0x80000040);
|
||||||
|
EXPECT_EQ(core.RunCycles(1), 1u);
|
||||||
|
|
||||||
|
const u64 dispatches_before_revalidation = core.GetJitDispatchSlowCount();
|
||||||
|
core.SetRegister(15, 0x80000000);
|
||||||
|
EXPECT_EQ(core.RunCycles(1), 1u);
|
||||||
|
EXPECT_EQ(core.GetJitDispatchSlowCount(), dispatches_before_revalidation);
|
||||||
|
|
||||||
|
// The first block revalidated the unchanged page-table descriptor directly in generated code.
|
||||||
|
// A second native block on that physical page must likewise avoid a page-table walk and C++
|
||||||
|
// block-map lookup.
|
||||||
|
core.SetRegister(15, 0x80000020);
|
||||||
|
EXPECT_EQ(core.RunCycles(1), 1u);
|
||||||
|
EXPECT_EQ(core.GetJitDispatchSlowCount(), dispatches_before_revalidation);
|
||||||
|
#endif
|
||||||
|
}
|
||||||
|
|
||||||
|
TEST(StarletARMCore, JitSharedTLBRefillRejectsRemappedPhysicalCode)
|
||||||
|
{
|
||||||
|
#if defined(_M_X86_64)
|
||||||
|
TestBus bus(0x110000);
|
||||||
|
ARMCore core(bus);
|
||||||
|
bus.WriteARM(0x000000, 0xe3a01001); // old page: mov r1, #1
|
||||||
|
bus.WriteARM(0x000020, 0xe3a02002); // old page: mov r2, #2
|
||||||
|
bus.WriteARM(0x000040, 0xee080f17); // invalidate unified TLB
|
||||||
|
bus.WriteARM(0x100000, 0xe3a01003); // new page: mov r1, #3
|
||||||
|
bus.WriteARM(0x100020, 0xe3a02004); // new page: mov r2, #4
|
||||||
|
bus.WriteARM(0x006000, 0x00000c02); // VA 0x80000000 section -> PA 0
|
||||||
|
core.GetCP15State().translation_table_base = 0x4000;
|
||||||
|
core.GetCP15State().domain_access_control = 3;
|
||||||
|
core.GetCP15State().control |= 1;
|
||||||
|
core.SetJitEnabled(true);
|
||||||
|
|
||||||
|
core.SetRegister(15, 0x80000000);
|
||||||
|
EXPECT_EQ(core.RunCycles(1), 1u);
|
||||||
|
core.SetRegister(15, 0x80000020);
|
||||||
|
EXPECT_EQ(core.RunCycles(1), 1u);
|
||||||
|
const size_t old_block_count = core.GetJitCompiledBlockCount();
|
||||||
|
|
||||||
|
// Change the page table under the still-valid TLB, then execute the architectural invalidation
|
||||||
|
// through the old mapping. The following dispatch must discover the new physical page.
|
||||||
|
bus.WriteARM(0x006000, 0x00100c02);
|
||||||
|
core.SetRegister(15, 0x80000040);
|
||||||
|
EXPECT_EQ(core.RunCycles(1), 1u);
|
||||||
|
core.SetRegister(15, 0x80000000);
|
||||||
|
EXPECT_EQ(core.RunCycles(1), 1u);
|
||||||
|
EXPECT_EQ(core.GetRegister(1), 3u);
|
||||||
|
core.SetRegister(15, 0x80000020);
|
||||||
|
EXPECT_EQ(core.RunCycles(1), 1u);
|
||||||
|
EXPECT_EQ(core.GetRegister(2), 4u);
|
||||||
|
EXPECT_EQ(core.GetJitCompiledBlockCount(), old_block_count + 3);
|
||||||
|
#endif
|
||||||
|
}
|
||||||
|
|
||||||
|
TEST(StarletARMCore, JitRetainsPhysicalAliasesAcrossAddressSpaceSwitches)
|
||||||
|
{
|
||||||
|
#if defined(_M_X86_64)
|
||||||
|
TestBus bus(0x110000);
|
||||||
|
ARMCore core(bus);
|
||||||
|
bus.WriteARM(0x000000, 0xe3a01001); // physical mapping 0: mov r1, #1
|
||||||
|
bus.WriteARM(0x000020, 0xe3a02002); // physical mapping 0: mov r2, #2
|
||||||
|
bus.WriteARM(0x000040, 0xee080f17); // invalidate unified TLB
|
||||||
|
bus.WriteARM(0x100000, 0xe3a01003); // physical mapping 1: mov r1, #3
|
||||||
|
bus.WriteARM(0x100020, 0xe3a02004); // physical mapping 1: mov r2, #4
|
||||||
|
bus.WriteARM(0x100040, 0xee080f17); // invalidate unified TLB
|
||||||
|
bus.WriteARM(0x006000, 0x00000c02); // VA 0x80000000 section -> PA 0
|
||||||
|
core.GetCP15State().translation_table_base = 0x4000;
|
||||||
|
core.GetCP15State().domain_access_control = 3;
|
||||||
|
core.GetCP15State().control |= 1;
|
||||||
|
core.SetJitEnabled(true);
|
||||||
|
|
||||||
|
for (const u32 address : {0x80000000U, 0x80000020U})
|
||||||
|
{
|
||||||
|
core.SetRegister(15, address);
|
||||||
|
EXPECT_EQ(core.RunCycles(1), 1u);
|
||||||
|
}
|
||||||
|
|
||||||
|
bus.WriteARM(0x006000, 0x00100c02); // Same virtual section -> PA 1 MiB.
|
||||||
|
core.SetRegister(15, 0x80000040);
|
||||||
|
EXPECT_EQ(core.RunCycles(1), 1u);
|
||||||
|
for (const u32 address : {0x80000000U, 0x80000020U})
|
||||||
|
{
|
||||||
|
core.SetRegister(15, address);
|
||||||
|
EXPECT_EQ(core.RunCycles(1), 1u);
|
||||||
|
}
|
||||||
|
|
||||||
|
// Return to the first address space. The first block refills the shared page translation; the
|
||||||
|
// second must immediately find its retained (MVA, physical page) entry instead of overwriting a
|
||||||
|
// single virtual-key slot and falling back to C++ again.
|
||||||
|
bus.WriteARM(0x006000, 0x00000c02);
|
||||||
|
core.SetRegister(15, 0x80000040);
|
||||||
|
EXPECT_EQ(core.RunCycles(1), 1u);
|
||||||
|
const u64 dispatches_before_refill = core.GetJitDispatchSlowCount();
|
||||||
|
core.SetRegister(15, 0x80000000);
|
||||||
|
EXPECT_EQ(core.RunCycles(1), 1u);
|
||||||
|
EXPECT_EQ(core.GetRegister(1), 1u);
|
||||||
|
core.SetRegister(15, 0x80000020);
|
||||||
|
EXPECT_EQ(core.RunCycles(1), 1u);
|
||||||
|
EXPECT_EQ(core.GetRegister(2), 2u);
|
||||||
|
EXPECT_EQ(core.GetJitDispatchSlowCount(), dispatches_before_refill + 1);
|
||||||
|
#endif
|
||||||
|
}
|
||||||
|
|
||||||
|
TEST(StarletARMCore, JitFastBlockCacheRetainsFourCollidingHotBlocks)
|
||||||
|
{
|
||||||
|
#if defined(_M_X86_64)
|
||||||
|
TestBus bus(0xd0000);
|
||||||
|
ARMCore core(bus);
|
||||||
|
core.SetJitEnabled(true);
|
||||||
|
|
||||||
|
// These ARM addresses deliberately have the same upper 16 bits after multiplying by the JIT
|
||||||
|
// cache's 0x9e3779b1 hash constant. They therefore occupy the four ways of one cache set.
|
||||||
|
constexpr std::array<u32, 5> addresses = {0x000014, 0x04cb94, 0x07e168, 0x099714, 0x0cace8};
|
||||||
|
for (const u32 address : addresses)
|
||||||
|
bus.WriteARM(address, 0xe3a01001); // mov r1, #1
|
||||||
|
|
||||||
|
for (size_t i = 0; i < 4; ++i)
|
||||||
|
{
|
||||||
|
const u32 address = addresses[i];
|
||||||
|
core.SetRegister(15, address);
|
||||||
|
EXPECT_EQ(core.RunCycles(1), 1u);
|
||||||
|
}
|
||||||
|
|
||||||
|
const u64 dispatches_after_fill = core.GetJitDispatchSlowCount();
|
||||||
|
const u64 collisions_after_fill = core.GetJitDispatchCollisionCount();
|
||||||
|
for (size_t i = 0; i < 4; ++i)
|
||||||
|
{
|
||||||
|
const u32 address = addresses[i];
|
||||||
|
core.SetRegister(15, address);
|
||||||
|
EXPECT_EQ(core.RunCycles(1), 1u);
|
||||||
|
}
|
||||||
|
EXPECT_EQ(core.GetJitDispatchSlowCount(), dispatches_after_fill);
|
||||||
|
EXPECT_EQ(core.GetJitDispatchCollisionCount(), collisions_after_fill);
|
||||||
|
|
||||||
|
// A fifth distinct key proves that the set is actually full and exercises bounded replacement.
|
||||||
|
core.SetRegister(15, addresses.back());
|
||||||
|
EXPECT_EQ(core.RunCycles(1), 1u);
|
||||||
|
EXPECT_EQ(core.GetJitDispatchSlowCount(), dispatches_after_fill + 1);
|
||||||
|
EXPECT_EQ(core.GetJitDispatchCollisionCount(), collisions_after_fill + 1);
|
||||||
|
#endif
|
||||||
|
}
|
||||||
|
|
||||||
|
TEST(StarletARMCore, JitCachesFallbackOnlyBlocks)
|
||||||
|
{
|
||||||
|
#if defined(_M_X86_64)
|
||||||
|
TestBus bus(0x10000);
|
||||||
|
ARMCore core(bus);
|
||||||
|
// MUL uses the exact interpreter helper in this JIT. A block beginning with it therefore has
|
||||||
|
// zero directly emitted ARM instructions, but its generated fallback wrapper is still reusable.
|
||||||
|
bus.WriteARM(0x0000, 0xe0010190); // mul r1, r0, r1
|
||||||
|
core.SetJitEnabled(true);
|
||||||
|
|
||||||
|
core.SetRegister(0, 3);
|
||||||
|
core.SetRegister(1, 4);
|
||||||
|
core.SetRegister(15, 0);
|
||||||
|
EXPECT_EQ(core.RunCycles(1), 1u);
|
||||||
|
EXPECT_EQ(core.GetRegister(1), 12u);
|
||||||
|
const u64 dispatches_after_compile = core.GetJitDispatchSlowCount();
|
||||||
|
const u64 fallbacks_after_compile = core.GetJitFallbackInstructionCount();
|
||||||
|
|
||||||
|
core.SetRegister(1, 5);
|
||||||
|
core.SetRegister(15, 0);
|
||||||
|
EXPECT_EQ(core.RunCycles(1), 1u);
|
||||||
|
EXPECT_EQ(core.GetRegister(1), 15u);
|
||||||
|
EXPECT_EQ(core.GetJitFallbackInstructionCount(), fallbacks_after_compile + 1);
|
||||||
|
EXPECT_EQ(core.GetJitDispatchSlowCount(), dispatches_after_compile);
|
||||||
|
#endif
|
||||||
|
}
|
||||||
|
|
||||||
|
TEST(StarletARMCore, ARMJitCompilesLogicalImmediateAndShiftCarry)
|
||||||
|
{
|
||||||
|
#if defined(_M_X86_64)
|
||||||
|
TestBus interpreter_bus(0x1000);
|
||||||
|
TestBus jit_bus(0x1000);
|
||||||
|
ARMCore interpreter(interpreter_bus);
|
||||||
|
ARMCore jit(jit_bus);
|
||||||
|
jit.SetJitEnabled(true);
|
||||||
|
const auto install_program = [](TestBus& bus) {
|
||||||
|
bus.WriteARM(0x00, 0xe3180701); // tst r8, #0x40000; rotated immediate supplies C
|
||||||
|
bus.WriteARM(0x04, 0xeafffffe); // b .
|
||||||
|
bus.WriteARM(0x20, 0xe1b02820); // movs r2, r0, lsr #16; bit 15 supplies C
|
||||||
|
bus.WriteARM(0x24, 0xeafffffe); // b .
|
||||||
|
};
|
||||||
|
install_program(interpreter_bus);
|
||||||
|
install_program(jit_bus);
|
||||||
|
|
||||||
|
for (ARMCore* core : {&interpreter, &jit})
|
||||||
|
{
|
||||||
|
core->SetCPSR(static_cast<u32>(ARMCore::Mode::System) | ARMCore::CPSR_C | ARMCore::CPSR_V);
|
||||||
|
core->SetRegister(8, 0x40000);
|
||||||
|
core->SetRegister(15, 0);
|
||||||
|
}
|
||||||
|
ASSERT_EQ(interpreter.RunCycles(2), 2u);
|
||||||
|
ASSERT_EQ(jit.RunCycles(2), 2u);
|
||||||
|
EXPECT_EQ(jit.GetCPSR(), interpreter.GetCPSR());
|
||||||
|
|
||||||
|
for (ARMCore* core : {&interpreter, &jit})
|
||||||
|
{
|
||||||
|
core->SetCPSR(static_cast<u32>(ARMCore::Mode::System) | ARMCore::CPSR_V);
|
||||||
|
core->SetRegister(0, 0x80018000);
|
||||||
|
core->SetRegister(2, 0);
|
||||||
|
core->SetRegister(15, 0x20);
|
||||||
|
}
|
||||||
|
ASSERT_EQ(interpreter.RunCycles(2), 2u);
|
||||||
|
ASSERT_EQ(jit.RunCycles(2), 2u);
|
||||||
|
EXPECT_EQ(jit.GetRegister(2), interpreter.GetRegister(2));
|
||||||
|
EXPECT_EQ(jit.GetCPSR(), interpreter.GetCPSR());
|
||||||
|
EXPECT_EQ(jit.GetJitFallbackInstructionCount(), 0u);
|
||||||
|
EXPECT_EQ(jit.GetJitNativeExecutedInstructions(), 4u);
|
||||||
|
#endif
|
||||||
|
}
|
||||||
|
|
||||||
|
TEST(StarletARMCore, ARMJitCompilesIRQVectorLoadPCWithInterworking)
|
||||||
|
{
|
||||||
|
#if defined(_M_X86_64)
|
||||||
|
TestBus bus;
|
||||||
|
ARMCore core(bus);
|
||||||
|
bus.SetFastmemEnabled(true);
|
||||||
|
bus.WriteARM(0x00, 0xe59ff018); // ldr pc, [pc, #0x18] -> 0x20
|
||||||
|
bus.WriteARM(0x20, 0x00000101); // enter Thumb at 0x100
|
||||||
|
core.SetRegister(15, 0);
|
||||||
|
core.SetJitEnabled(true);
|
||||||
|
|
||||||
|
EXPECT_EQ(core.RunCycles(1), 1u);
|
||||||
|
EXPECT_EQ(core.GetRegister(15), 0x100u);
|
||||||
|
EXPECT_NE(core.GetCPSR() & ARMCore::CPSR_T, 0u);
|
||||||
|
EXPECT_EQ(core.GetJitFallbackInstructionCount(), 0u);
|
||||||
|
EXPECT_EQ(core.GetJitSlowReadCount(), 0u);
|
||||||
|
EXPECT_EQ(core.GetJitNativeExecutedInstructions(), 1u);
|
||||||
|
EXPECT_TRUE(bus.SRAMCanariesIntact());
|
||||||
|
#endif
|
||||||
|
}
|
||||||
|
|
||||||
|
TEST(StarletARMCore, ARMJitCompilesLongMultiplyFamily)
|
||||||
|
{
|
||||||
|
#if defined(_M_X86_64)
|
||||||
|
struct Case
|
||||||
|
{
|
||||||
|
u32 instruction;
|
||||||
|
u32 cpsr;
|
||||||
|
u32 rm;
|
||||||
|
u32 rs;
|
||||||
|
u32 rd_hi;
|
||||||
|
u32 rd_lo;
|
||||||
|
};
|
||||||
|
constexpr std::array cases = {
|
||||||
|
Case{0xe0834291, static_cast<u32>(ARMCore::Mode::System) | ARMCore::CPSR_C, 0x10000, 0x10001,
|
||||||
|
0, 0}, // UMULL
|
||||||
|
Case{0xe0c34291, static_cast<u32>(ARMCore::Mode::System) | ARMCore::CPSR_V, 0xfffffff0, 0x10,
|
||||||
|
0, 0}, // SMULL
|
||||||
|
Case{0xe0b34291, static_cast<u32>(ARMCore::Mode::System) | ARMCore::CPSR_C | ARMCore::CPSR_V,
|
||||||
|
0xffffffff, 1, 0, 1}, // UMLALS, result wraps to zero and preserves CV
|
||||||
|
Case{0x10834291, static_cast<u32>(ARMCore::Mode::System) | ARMCore::CPSR_Z, 7, 9, 0x11223344,
|
||||||
|
0x55667788}, // UMULLNE, predicate fails
|
||||||
|
};
|
||||||
|
|
||||||
|
for (const Case& test : cases)
|
||||||
|
{
|
||||||
|
TestBus interpreter_bus;
|
||||||
|
TestBus jit_bus;
|
||||||
|
ARMCore interpreter(interpreter_bus);
|
||||||
|
ARMCore jit(jit_bus);
|
||||||
|
interpreter_bus.WriteARM(0, test.instruction);
|
||||||
|
interpreter_bus.WriteARM(4, 0xeafffffe); // b .
|
||||||
|
jit_bus.WriteARM(0, test.instruction);
|
||||||
|
jit_bus.WriteARM(4, 0xeafffffe); // b .
|
||||||
|
for (ARMCore* core : {&interpreter, &jit})
|
||||||
|
{
|
||||||
|
core->SetCPSR(test.cpsr);
|
||||||
|
core->SetRegister(1, test.rm);
|
||||||
|
core->SetRegister(2, test.rs);
|
||||||
|
core->SetRegister(3, test.rd_hi);
|
||||||
|
core->SetRegister(4, test.rd_lo);
|
||||||
|
core->SetRegister(15, 0);
|
||||||
|
}
|
||||||
|
jit.SetJitEnabled(true);
|
||||||
|
|
||||||
|
ASSERT_EQ(interpreter.RunCycles(2), 2u);
|
||||||
|
ASSERT_EQ(jit.RunCycles(2), 2u);
|
||||||
|
EXPECT_EQ(jit.GetRegister(3), interpreter.GetRegister(3));
|
||||||
|
EXPECT_EQ(jit.GetRegister(4), interpreter.GetRegister(4));
|
||||||
|
EXPECT_EQ(jit.GetCPSR(), interpreter.GetCPSR());
|
||||||
|
EXPECT_EQ(jit.GetJitFallbackInstructionCount(), 0u);
|
||||||
|
EXPECT_EQ(jit.GetJitNativeExecutedInstructions(), 2u);
|
||||||
|
}
|
||||||
|
#endif
|
||||||
|
}
|
||||||
|
|
||||||
TEST(StarletARMCore, FCSESwitchPreservesTaggedTLBTranslations)
|
TEST(StarletARMCore, FCSESwitchPreservesTaggedTLBTranslations)
|
||||||
{
|
{
|
||||||
TestBus bus(0x10000);
|
TestBus bus(0x10000);
|
||||||
|
|||||||
@@ -0,0 +1,316 @@
|
|||||||
|
// Copyright 2026 Dolphin Emulator Project
|
||||||
|
// SPDX-License-Identifier: GPL-2.0-or-later
|
||||||
|
|
||||||
|
#include <array>
|
||||||
|
#include <memory>
|
||||||
|
|
||||||
|
#include "Common/x64ABI.h"
|
||||||
|
#include "Core/ConfigManager.h"
|
||||||
|
#include "Core/Core.h"
|
||||||
|
#include "Core/HW/Memmap.h"
|
||||||
|
#include "Core/PowerPC/Jit64/Jit.h"
|
||||||
|
#include "Core/PowerPC/Jit64Common/Jit64AsmCommon.h"
|
||||||
|
#include "Core/PowerPC/Jit64Common/Jit64Constants.h"
|
||||||
|
#include "Core/PowerPC/MMU.h"
|
||||||
|
#include "Core/PowerPC/PowerPC.h"
|
||||||
|
#include "Core/System.h"
|
||||||
|
|
||||||
|
#include <gtest/gtest.h>
|
||||||
|
|
||||||
|
namespace
|
||||||
|
{
|
||||||
|
using namespace Gen;
|
||||||
|
|
||||||
|
// Execute the real SafeLoad/SafeWrite emitter, including its fallback and register contract.
|
||||||
|
// Testing only the C++ MMU helpers cannot detect corruption introduced at the native call site.
|
||||||
|
class DCacheCode : public CommonAsmRoutines
|
||||||
|
{
|
||||||
|
public:
|
||||||
|
explicit DCacheCode(Core::System& system) : CommonAsmRoutines(jit), jit(system)
|
||||||
|
{
|
||||||
|
jit.jo = {};
|
||||||
|
jit.js = {};
|
||||||
|
AllocCodeSpace(512 * 1024);
|
||||||
|
old_read = dcache32_read_hit_dbat;
|
||||||
|
old_write = dcache32_write_hit_dbat;
|
||||||
|
dcache32_read_hit_dbat = AlignCode4();
|
||||||
|
GenDCache32Hit(false);
|
||||||
|
dcache32_write_hit_dbat = AlignCode4();
|
||||||
|
GenDCache32Hit(true);
|
||||||
|
}
|
||||||
|
|
||||||
|
~DCacheCode() override
|
||||||
|
{
|
||||||
|
dcache32_read_hit_dbat = old_read;
|
||||||
|
dcache32_write_hit_dbat = old_write;
|
||||||
|
}
|
||||||
|
|
||||||
|
void Access(bool write, u32 address, u32 value, X64Reg address_reg, X64Reg value_reg,
|
||||||
|
s32 offset = 0, int flags = 0)
|
||||||
|
{
|
||||||
|
run = reinterpret_cast<void (*)()>(AlignCode4());
|
||||||
|
ABI_PushRegistersAndAdjustStack(ABI_ALL_CALLEE_SAVED, 8);
|
||||||
|
MOV(64, R(RPPCSTATE), ImmPtr(reinterpret_cast<u8*>(&jit.m_ppc_state) + 0x80));
|
||||||
|
MOV(32, R(RDX), Imm32(0x12345678));
|
||||||
|
MOV(32, R(RCX), Imm32(0x87654321));
|
||||||
|
MOV(32, R(address_reg), Imm32(address));
|
||||||
|
if (write)
|
||||||
|
MOV(32, R(value_reg), Imm32(value));
|
||||||
|
const BitSet32 live{RCX, RDX, R8, R9};
|
||||||
|
if (flags & SAFE_LOADSTORE_NO_PROLOG)
|
||||||
|
SUB(64, R(RSP), Imm8(8));
|
||||||
|
if (write)
|
||||||
|
SafeWriteRegToReg(value_reg, address_reg, 32, offset, live, flags);
|
||||||
|
else
|
||||||
|
SafeLoadToReg(value_reg, R(address_reg), 32, offset, live, false, flags);
|
||||||
|
if (flags & SAFE_LOADSTORE_NO_PROLOG)
|
||||||
|
ADD(64, R(RSP), Imm8(8));
|
||||||
|
|
||||||
|
MOV(64, R(R11), ImmPtr(result.data()));
|
||||||
|
for (const auto reg : {RAX, RCX, RDX, R8, R9})
|
||||||
|
MOV(64, MDisp(R11, static_cast<int>(reg) * sizeof(u64)), R(reg));
|
||||||
|
ABI_PopRegistersAndAdjustStack(ABI_ALL_CALLEE_SAVED, 8);
|
||||||
|
RET();
|
||||||
|
ASSERT_FALSE(HasWriteFailed());
|
||||||
|
run();
|
||||||
|
}
|
||||||
|
|
||||||
|
std::array<u64, 16> result{};
|
||||||
|
void (*run)() = nullptr;
|
||||||
|
Jit64 jit;
|
||||||
|
const u8* old_read;
|
||||||
|
const u8* old_write;
|
||||||
|
};
|
||||||
|
|
||||||
|
class Jit64DCache : public testing::Test
|
||||||
|
{
|
||||||
|
protected:
|
||||||
|
static void SetUpTestSuite()
|
||||||
|
{
|
||||||
|
SConfig::Init();
|
||||||
|
auto& system = Core::System::GetInstance();
|
||||||
|
system.SetIsWii(true);
|
||||||
|
system.GetMemory().Init();
|
||||||
|
Core::DeclareAsCPUThread();
|
||||||
|
system.GetPPCState().dCache.Init(system.GetMemory());
|
||||||
|
}
|
||||||
|
|
||||||
|
static void TearDownTestSuite()
|
||||||
|
{
|
||||||
|
auto& system = Core::System::GetInstance();
|
||||||
|
system.GetPPCState().m_enable_dcache = false;
|
||||||
|
system.GetMemory().Shutdown();
|
||||||
|
system.SetIsWii(false);
|
||||||
|
Core::UndeclareAsCPUThread();
|
||||||
|
SConfig::Shutdown();
|
||||||
|
}
|
||||||
|
|
||||||
|
void SetUp() override
|
||||||
|
{
|
||||||
|
state.dCache.Reset();
|
||||||
|
state.m_enable_dcache = true;
|
||||||
|
state.msr.DR = 1;
|
||||||
|
state.feature_flags = FEATURE_FLAG_MSR_DR;
|
||||||
|
state.Exceptions = 0;
|
||||||
|
state.spr[SPR_HID0] = 0;
|
||||||
|
bats.fill(0);
|
||||||
|
bats[0x80000000 >> PowerPC::BAT_INDEX_SHIFT] = PowerPC::BAT_MAPPED_BIT;
|
||||||
|
bats[0x90000000 >> PowerPC::BAT_INDEX_SHIFT] = 0x10000000 | PowerPC::BAT_MAPPED_BIT;
|
||||||
|
code = std::make_unique<DCacheCode>(system);
|
||||||
|
}
|
||||||
|
|
||||||
|
void TearDown() override { code.reset(); }
|
||||||
|
|
||||||
|
Core::System& system = Core::System::GetInstance();
|
||||||
|
PowerPC::PowerPCState& state = system.GetPPCState();
|
||||||
|
PowerPC::MMU& mmu = system.GetMMU();
|
||||||
|
Memory::MemoryManager& memory = system.GetMemory();
|
||||||
|
PowerPC::BatTable& bats = const_cast<PowerPC::BatTable&>(mmu.GetDBATTable());
|
||||||
|
std::unique_ptr<DCacheCode> code;
|
||||||
|
};
|
||||||
|
|
||||||
|
TEST_F(Jit64DCache, StoreMissRetainsAddressInRDX)
|
||||||
|
{
|
||||||
|
// The data is deliberately also a mapped address: a broken fallback writes there instead.
|
||||||
|
memory.Write_U32(0, 0x1000);
|
||||||
|
memory.Write_U32(0, 0x4000);
|
||||||
|
code->Access(true, 0x80001000, 0x80004000, RDX, RCX);
|
||||||
|
EXPECT_EQ(mmu.ReadForJit<u32>(0x80001000), 0x80004000);
|
||||||
|
EXPECT_EQ(mmu.ReadForJit<u32>(0x80004000), 0U);
|
||||||
|
EXPECT_EQ(code->result[RDX], 0x80001000);
|
||||||
|
EXPECT_EQ(code->result[RCX], 0x80004000);
|
||||||
|
}
|
||||||
|
|
||||||
|
TEST_F(Jit64DCache, LoadMissRetainsAddressInRDX)
|
||||||
|
{
|
||||||
|
memory.Write_U32(0x89abcdef, 0x1000);
|
||||||
|
code->Access(false, 0x80001000, 0, RDX, R8);
|
||||||
|
EXPECT_EQ(code->result[R8], 0x89abcdef);
|
||||||
|
EXPECT_EQ(code->result[RDX], 0x80001000);
|
||||||
|
}
|
||||||
|
|
||||||
|
TEST_F(Jit64DCache, HitPreservesLiveScratchRegisters)
|
||||||
|
{
|
||||||
|
mmu.WriteForJit<u32>(0xffffffff, 0x80001000);
|
||||||
|
code->Access(false, 0x80001000, 0, R9, R8);
|
||||||
|
EXPECT_EQ(code->result[R8], 0xffffffff);
|
||||||
|
EXPECT_EQ(code->result[RDX], 0x12345678U);
|
||||||
|
EXPECT_EQ(code->result[RCX], 0x87654321U);
|
||||||
|
code->Access(true, 0x80001000, 0x11223344, R9, R8);
|
||||||
|
EXPECT_EQ(mmu.ReadForJit<u32>(0x80001000), 0x11223344U);
|
||||||
|
EXPECT_EQ(code->result[RDX], 0x12345678U);
|
||||||
|
EXPECT_EQ(code->result[RCX], 0x87654321U);
|
||||||
|
}
|
||||||
|
|
||||||
|
TEST_F(Jit64DCache, PhysicalAccessIgnoresDBAT)
|
||||||
|
{
|
||||||
|
state.msr.DR = 0;
|
||||||
|
state.feature_flags = CPUEmuFeatureFlags{};
|
||||||
|
bats[0] = 0x20000 | PowerPC::BAT_MAPPED_BIT;
|
||||||
|
mmu.WriteForJit<u32>(0x11111111, 0x1000);
|
||||||
|
mmu.WriteForJit<u32>(0x22222222, 0x21000);
|
||||||
|
code->Access(false, 0x1000, 0, R9, R8);
|
||||||
|
EXPECT_EQ(code->result[R8], 0x11111111U);
|
||||||
|
code->Access(true, 0x1000, 0x33333333, R9, R8);
|
||||||
|
EXPECT_EQ(mmu.ReadForJit<u32>(0x1000), 0x33333333U);
|
||||||
|
EXPECT_EQ(mmu.ReadForJit<u32>(0x21000), 0x22222222U);
|
||||||
|
}
|
||||||
|
|
||||||
|
TEST_F(Jit64DCache, LoadRegisterAndOffsetCombinations)
|
||||||
|
{
|
||||||
|
for (const bool hit : {false, true})
|
||||||
|
{
|
||||||
|
for (const auto address_reg : {RAX, RCX, RDX, R8, R9})
|
||||||
|
{
|
||||||
|
for (const auto value_reg : {RAX, RCX, RDX, R8, R9})
|
||||||
|
{
|
||||||
|
for (const s32 offset : {0, 4, -4})
|
||||||
|
{
|
||||||
|
SCOPED_TRACE(testing::Message()
|
||||||
|
<< hit << ' ' << address_reg << ' ' << value_reg << ' ' << offset);
|
||||||
|
state.dCache.Reset();
|
||||||
|
memory.Write_U32(0xfedcba98, 0x1020);
|
||||||
|
if (hit)
|
||||||
|
ASSERT_EQ(mmu.ReadForJit<u32>(0x80001020), 0xfedcba98);
|
||||||
|
code->Access(false, 0x80001020 - offset, 0, address_reg, value_reg, offset);
|
||||||
|
EXPECT_EQ(code->result[value_reg], 0xfedcba98);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
TEST_F(Jit64DCache, StoreRegisterAndOffsetCombinations)
|
||||||
|
{
|
||||||
|
for (const bool hit : {false, true})
|
||||||
|
{
|
||||||
|
for (const auto address_reg : {RAX, RCX, RDX, R8, R9})
|
||||||
|
{
|
||||||
|
for (const auto value_reg : {RCX, RDX, R8, R9})
|
||||||
|
{
|
||||||
|
if (address_reg == value_reg)
|
||||||
|
continue;
|
||||||
|
for (const s32 offset : {0, 4, -4})
|
||||||
|
{
|
||||||
|
SCOPED_TRACE(testing::Message()
|
||||||
|
<< hit << ' ' << address_reg << ' ' << value_reg << ' ' << offset);
|
||||||
|
state.dCache.Reset();
|
||||||
|
memory.Write_U32(0, 0x1020);
|
||||||
|
if (hit)
|
||||||
|
ASSERT_EQ(mmu.ReadForJit<u32>(0x80001020), 0U);
|
||||||
|
code->Access(true, 0x80001020 - offset, 0xfedcba98, address_reg, value_reg, offset,
|
||||||
|
EmuCodeBlock::SAFE_LOADSTORE_CLOBBER_RSCRATCH_INSTEAD_OF_ADDR);
|
||||||
|
EXPECT_EQ(mmu.ReadForJit<u32>(0x80001020), 0xfedcba98);
|
||||||
|
EXPECT_EQ(code->result[value_reg], 0xfedcba98);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
TEST_F(Jit64DCache, SplitLineAndPageAccess)
|
||||||
|
{
|
||||||
|
for (const u32 offset : {29U, 30U, 31U, 4093U, 4094U, 4095U})
|
||||||
|
{
|
||||||
|
SCOPED_TRACE(offset);
|
||||||
|
const u32 address = 0x80001000 + offset;
|
||||||
|
mmu.WriteForJit<u32>(0x11223344, address);
|
||||||
|
code->Access(false, address, 0, RDX, R8);
|
||||||
|
EXPECT_EQ(code->result[R8], 0x11223344U);
|
||||||
|
code->Access(true, address, 0x55667788, RDX, RCX);
|
||||||
|
EXPECT_EQ(mmu.ReadForJit<u32>(address), 0x55667788U);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
TEST_F(Jit64DCache, InhibitedBATBypassesCachedData)
|
||||||
|
{
|
||||||
|
mmu.WriteForJit<u32>(0x11111111, 0x80001000);
|
||||||
|
memory.Write_U32(0x22222222, 0x1000);
|
||||||
|
bats[0x80000000 >> PowerPC::BAT_INDEX_SHIFT] |= PowerPC::BAT_WI_BIT;
|
||||||
|
code->Access(false, 0x80001000, 0, RDX, R8);
|
||||||
|
EXPECT_EQ(code->result[R8], 0x22222222U);
|
||||||
|
code->Access(true, 0x80001000, 0x33333333, RDX, RCX);
|
||||||
|
EXPECT_EQ(memory.Read_U32(0x1000), 0x33333333U);
|
||||||
|
bats[0x80000000 >> PowerPC::BAT_INDEX_SHIFT] &= ~PowerPC::BAT_WI_BIT;
|
||||||
|
EXPECT_EQ(mmu.ReadForJit<u32>(0x80001000), 0x11111111U);
|
||||||
|
}
|
||||||
|
|
||||||
|
TEST_F(Jit64DCache, SharedRoutineStackAndDRFlag)
|
||||||
|
{
|
||||||
|
// Shared paired-load routines are emitted without a known block's feature flags and run
|
||||||
|
// with a return address on the stack. DR_ON is their explicit translation contract.
|
||||||
|
state.feature_flags = CPUEmuFeatureFlags{};
|
||||||
|
constexpr int flags = EmuCodeBlock::SAFE_LOADSTORE_NO_PROLOG |
|
||||||
|
EmuCodeBlock::SAFE_LOADSTORE_NO_UPDATE_PC |
|
||||||
|
EmuCodeBlock::SAFE_LOADSTORE_DR_ON;
|
||||||
|
memory.Write_U32(0x89abcdef, 0x1000);
|
||||||
|
for (int repeat = 0; repeat < 2; ++repeat)
|
||||||
|
{
|
||||||
|
code->Access(false, 0x80001000, 0, RDX, R8, 0, flags);
|
||||||
|
EXPECT_EQ(code->result[R8], 0x89abcdef);
|
||||||
|
code->Access(true, 0x80001000, 0x89abcdef, RDX, RCX, 0, flags);
|
||||||
|
EXPECT_EQ(mmu.ReadForJit<u32>(0x80001000), 0x89abcdef);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
TEST_F(Jit64DCache, NativePLRUAndDirtyStateMatchMMU)
|
||||||
|
{
|
||||||
|
for (const u32 base : {0x80000000U, 0x90000000U})
|
||||||
|
{
|
||||||
|
state.dCache.Reset();
|
||||||
|
constexpr u32 set = 126;
|
||||||
|
for (u32 way = 0; way < PowerPC::CACHE_WAYS; ++way)
|
||||||
|
mmu.WriteForJit<u32>(0xffffffff, base + way * 4096 + set * 32);
|
||||||
|
for (u32 way = 0; way < PowerPC::CACHE_WAYS; ++way)
|
||||||
|
{
|
||||||
|
const u32 address = base + way * 4096 + set * 32;
|
||||||
|
for (const bool write : {false, true})
|
||||||
|
{
|
||||||
|
code->Access(write, address, 0xffffffff, RDX, R8);
|
||||||
|
for (u32 old_plru = 0; old_plru < 128; ++old_plru)
|
||||||
|
{
|
||||||
|
SCOPED_TRACE(testing::Message() << base << ' ' << way << ' ' << write << ' ' << old_plru);
|
||||||
|
auto& cache = state.dCache;
|
||||||
|
cache.plru[set] = static_cast<u8>(old_plru);
|
||||||
|
cache.modified[set] = 0x80;
|
||||||
|
if (write)
|
||||||
|
ASSERT_TRUE(mmu.TryWriteDCache32ForJit(0xffffffff, address));
|
||||||
|
else
|
||||||
|
ASSERT_EQ(mmu.TryReadDCache32ForJit(address), 0x100000000ULL);
|
||||||
|
const auto expected_plru = cache.plru[set];
|
||||||
|
const auto expected_dirty = cache.modified[set];
|
||||||
|
const auto expected_data = cache.data[set];
|
||||||
|
cache.plru[set] = static_cast<u8>(old_plru);
|
||||||
|
cache.modified[set] = 0x80;
|
||||||
|
code->run();
|
||||||
|
EXPECT_EQ(cache.plru[set], expected_plru);
|
||||||
|
EXPECT_EQ(cache.modified[set], expected_dirty);
|
||||||
|
EXPECT_EQ(cache.data[set], expected_data);
|
||||||
|
if (!write)
|
||||||
|
EXPECT_EQ(code->result[R8], 0xffffffff);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
} // namespace
|
||||||
Reference in New Issue
Block a user