IOS: checkpoint Starlet JIT, PPC cache and Wii timing fixes

Save the remaining ARM JIT and MMU/cache optimizations, accurate Starlet timer and Wiimote report cadence, opt-in PPC event tracing, and full texture hashing in the LLE launcher. Include regression coverage and exclude local profiling artifacts.

Validation: 127 targeted tests from 15 suites passed, with one disabled test. Includes the current user-tested source state following the persistent NAND milestone.
This commit is contained in:
2026-09-08 11:45:51 +02:00
parent 31760e8bce
commit ebb753d09a
25 changed files with 2661 additions and 125 deletions
+5
View File
@@ -66,3 +66,8 @@ CMakeUserPresets.json
/capstone_local/
/capstone_runtime/
/capstone-*.whl
# Local profiling helpers, captures, and diagnostic output
/.codex-tools/
/.starlet-profile/
/yaya48-starlet-log.txt
+13 -2
View File
@@ -5,6 +5,8 @@ param(
[switch]$DisableJIT,
[switch]$SafeTextureCache = $true,
[switch]$Wait
)
@@ -16,7 +18,7 @@ if (-not (Test-Path -LiteralPath $dolphinPath)) {
throw "DolphinNoGUI.exe is missing: $dolphinPath"
}
$dolphin = Start-Process -FilePath $dolphinPath -ArgumentList @(
$dolphinArguments = @(
'-u', $userPath,
'-n', '0000000100000002',
'-C', "Dolphin.Core.WiiStarletJIT=$(-not $DisableJIT)",
@@ -27,7 +29,16 @@ $dolphin = Start-Process -FilePath $dolphinPath -ArgumentList @(
'-C', 'Logger.Logs.IOS=True',
'-v', 'D3D',
'-p', 'win32'
) -WorkingDirectory $repoRoot -PassThru
)
if ($SafeTextureCache) {
# Hash every texture byte: sparse samples can miss small CPU-rendered text updates.
# This is a per-run override and does not change the saved graphics configuration.
$dolphinArguments += @('-C', 'Graphics.Settings.SafeTextureCacheColorSamples=0')
}
$dolphin = Start-Process -FilePath $dolphinPath -ArgumentList $dolphinArguments `
-WorkingDirectory $repoRoot -PassThru
$dolphin.PriorityClass = 'High'
+1 -1
View File
@@ -317,7 +317,7 @@ void VideoInterfaceManager::RegisterMMIO(MMIO::Mapping* mmio, u32 base)
mmio->Register(
base | VI_VERTICAL_BEAM_POSITION, MMIO::ComplexRead<u16>([](Core::System& system, u32) {
auto& vi = system.GetVideoInterface();
return 1 + (vi.m_half_line_count) / 2;
return vi.GetVerticalBeamPosition();
}),
MMIO::ComplexWrite<u16>([](Core::System& system, u32, u16 val) {
auto& vi = system.GetVideoInterface();
+3
View File
@@ -389,6 +389,9 @@ public:
u32 GetTicksPerHalfLine() const;
u32 GetTicksPerField() const;
// Current one-based vertical beam position exposed by VI_VERTICAL_BEAM_POSITION.
u16 GetVerticalBeamPosition() const { return static_cast<u16>(1 + m_half_line_count / 2); }
// Not adjusted by VBI Clock Override.
u32 GetNominalTicksPerHalfLine() const;
+282 -12
View File
@@ -7,6 +7,7 @@
#include <array>
#include <bit>
#include <cassert>
#include <cstring>
#include <limits>
#include <utility>
@@ -201,43 +202,95 @@ u32 ARMCore::TranslateVirtualAddress(u32 address) const
return physical_address;
};
return cache_translation(WalkPageTables(modified_address, nullptr, nullptr, nullptr, nullptr));
}
u32 ARMCore::TranslateVirtualAddressForJit(u32 address, const u8** first_descriptor,
u32* first_descriptor_value,
const u8** second_descriptor,
u32* second_descriptor_value) const
{
*first_descriptor = nullptr;
*first_descriptor_value = 0;
*second_descriptor = nullptr;
*second_descriptor_value = 0;
if ((m_cp15.control & CP15_CONTROL_MMU) == 0)
return address;
const u32 modified_address =
address < 0x02000000 ? address | (m_cp15.process_id & 0xfe000000) : address;
const u32 physical_address =
WalkPageTables(modified_address, first_descriptor, first_descriptor_value, second_descriptor,
second_descriptor_value);
// Keep the interpreter and generated data-access path coherent with the translation established
// for this native code block.
const u32 virtual_page = modified_address >> 10;
TLBEntry& entry = m_tlb[GetTLBIndex(virtual_page)];
entry.virtual_page = virtual_page;
entry.physical_page = physical_address & ~0x3ffU;
entry.generation = m_tlb_generation;
return physical_address;
}
u32 ARMCore::WalkPageTables(u32 modified_address, const u8** first_descriptor,
u32* first_descriptor_value, const u8** second_descriptor,
u32* second_descriptor_value) const
{
const auto read_descriptor = [&](u32 physical_address, const u8** host_pointer,
u32* host_value) {
const u8* const pointer = m_bus.GetDirectMemoryPointer(physical_address, sizeof(u32));
if (host_pointer)
*host_pointer = pointer;
if (host_value)
{
*host_value = 0;
if (pointer)
std::memcpy(host_value, pointer, sizeof(u32));
}
return ReadPhysical32(physical_address);
};
const u32 first_level_address =
(m_cp15.translation_table_base & 0xffffc000) | ((modified_address >> 18) & 0x3ffc);
const u32 first_level = ReadPhysical32(first_level_address);
const u32 first_level =
read_descriptor(first_level_address, first_descriptor, first_descriptor_value);
switch (first_level & 3)
{
case 1: // Coarse second-level table.
{
const u32 second_level_address =
(first_level & 0xfffffc00) | ((modified_address >> 10) & 0x3fc);
const u32 second_level = ReadPhysical32(second_level_address);
const u32 second_level =
read_descriptor(second_level_address, second_descriptor, second_descriptor_value);
switch (second_level & 3)
{
case 1: // 64 KiB large page.
return cache_translation((second_level & 0xffff0000) | (modified_address & 0xffff));
return (second_level & 0xffff0000) | (modified_address & 0xffff);
case 2:
case 3: // 4 KiB small page; extended small pages share this mapping shape.
return cache_translation((second_level & 0xfffff000) | (modified_address & 0xfff));
return (second_level & 0xfffff000) | (modified_address & 0xfff);
default:
return cache_translation(modified_address);
return modified_address;
}
}
case 2: // 1 MiB section.
return cache_translation((first_level & 0xfff00000) | (modified_address & 0x000fffff));
return (first_level & 0xfff00000) | (modified_address & 0x000fffff);
case 3: // Fine second-level table.
{
const u32 second_level_address = (first_level & 0xfffff000) | ((modified_address >> 8) & 0xffc);
const u32 second_level = ReadPhysical32(second_level_address);
const u32 second_level =
read_descriptor(second_level_address, second_descriptor, second_descriptor_value);
switch (second_level & 3)
{
case 1:
return cache_translation((second_level & 0xffff0000) | (modified_address & 0xffff));
return (second_level & 0xffff0000) | (modified_address & 0xffff);
case 2:
return cache_translation((second_level & 0xfffff000) | (modified_address & 0xfff));
return (second_level & 0xfffff000) | (modified_address & 0xfff);
case 3: // 1 KiB tiny page.
return cache_translation((second_level & 0xfffffc00) | (modified_address & 0x3ff));
return (second_level & 0xfffffc00) | (modified_address & 0x3ff);
default:
return cache_translation(modified_address);
return modified_address;
}
}
default:
@@ -246,7 +299,7 @@ u32 ARMCore::TranslateVirtualAddress(u32 address) const
// fallback keeps diagnostics observable meanwhile. Cache that provisional translation just
// like a mapped page: real software must invalidate the TLB after changing a page-table entry,
// and without this cache the JIT would side-exit forever on every identity access.
return cache_translation(modified_address);
return modified_address;
}
}
@@ -864,6 +917,213 @@ size_t ARMCore::GetJitCompiledBlockCount() const
#endif
}
u64 ARMCore::GetJitLifetimeCompiledBlockCount() const
{
#if defined(_M_X86_64)
return m_jit ? m_jit->GetLifetimeCompiledBlockCount() : 0;
#else
return 0;
#endif
}
u64 ARMCore::GetJitBlockLookupCount() const
{
#if defined(_M_X86_64)
return m_jit ? m_jit->GetBlockLookupCount() : 0;
#else
return 0;
#endif
}
u64 ARMCore::GetJitBlockCacheHitCount() const
{
#if defined(_M_X86_64)
return m_jit ? m_jit->GetBlockCacheHitCount() : 0;
#else
return 0;
#endif
}
u64 ARMCore::GetJitDispatchSlowCount() const
{
#if defined(_M_X86_64)
return m_jit ? m_jit->GetDispatchSlowCount() : 0;
#else
return 0;
#endif
}
u64 ARMCore::GetJitDispatchStaleGenerationCount() const
{
#if defined(_M_X86_64)
return m_jit ? m_jit->GetDispatchStaleGenerationCount() : 0;
#else
return 0;
#endif
}
u64 ARMCore::GetJitDispatchPhysicalAliasCount() const
{
#if defined(_M_X86_64)
return m_jit ? m_jit->GetDispatchPhysicalAliasCount() : 0;
#else
return 0;
#endif
}
u64 ARMCore::GetJitDispatchCollisionCount() const
{
#if defined(_M_X86_64)
return m_jit ? m_jit->GetDispatchCollisionCount() : 0;
#else
return 0;
#endif
}
u64 ARMCore::GetJitDispatchKeyMissCount() const
{
#if defined(_M_X86_64)
return m_jit ? m_jit->GetDispatchKeyMissCount() : 0;
#else
return 0;
#endif
}
u64 ARMCore::GetJitDispatchPageMissCount() const
{
#if defined(_M_X86_64)
return m_jit ? m_jit->GetDispatchPageMissCount() : 0;
#else
return 0;
#endif
}
u64 ARMCore::GetJitDispatchEmptyEntryCount() const
{
#if defined(_M_X86_64)
return m_jit ? m_jit->GetDispatchEmptyEntryCount() : 0;
#else
return 0;
#endif
}
u64 ARMCore::GetJitDispatchGeneratedKeyMismatchCount() const
{
#if defined(_M_X86_64)
return m_jit ? m_jit->GetDispatchGeneratedKeyMismatchCount() : 0;
#else
return 0;
#endif
}
u64 ARMCore::GetJitDispatchGeneratedSetMismatchCount() const
{
#if defined(_M_X86_64)
return m_jit ? m_jit->GetDispatchGeneratedSetMismatchCount() : 0;
#else
return 0;
#endif
}
u64 ARMCore::GetJitDispatchKeyMissWithExactMatchCount() const
{
#if defined(_M_X86_64)
return m_jit ? m_jit->GetDispatchKeyMissWithExactMatchCount() : 0;
#else
return 0;
#endif
}
u64 ARMCore::GetJitDispatchKeyMissWithPhysicalAliasCount() const
{
#if defined(_M_X86_64)
return m_jit ? m_jit->GetDispatchKeyMissWithPhysicalAliasCount() : 0;
#else
return 0;
#endif
}
u64 ARMCore::GetJitDispatchKeyMissWithSamePhysicalPageCount() const
{
#if defined(_M_X86_64)
return m_jit ? m_jit->GetDispatchKeyMissWithSamePhysicalPageCount() : 0;
#else
return 0;
#endif
}
u64 ARMCore::GetJitDispatchKeyMissWithEmptySlotCount() const
{
#if defined(_M_X86_64)
return m_jit ? m_jit->GetDispatchKeyMissWithEmptySlotCount() : 0;
#else
return 0;
#endif
}
u64 ARMCore::GetJitDispatchKeyMissWithFullSetCount() const
{
#if defined(_M_X86_64)
return m_jit ? m_jit->GetDispatchKeyMissWithFullSetCount() : 0;
#else
return 0;
#endif
}
u64 ARMCore::GetJitDispatchKeyMissWithStaleTranslationCount() const
{
#if defined(_M_X86_64)
return m_jit ? m_jit->GetDispatchKeyMissWithStaleTranslationCount() : 0;
#else
return 0;
#endif
}
u64 ARMCore::GetJitFastEntryEmptyInsertCount() const
{
#if defined(_M_X86_64)
return m_jit ? m_jit->GetFastEntryEmptyInsertCount() : 0;
#else
return 0;
#endif
}
u64 ARMCore::GetJitFastEntryReplacementCount() const
{
#if defined(_M_X86_64)
return m_jit ? m_jit->GetFastEntryReplacementCount() : 0;
#else
return 0;
#endif
}
u64 ARMCore::GetJitInstructionCacheClearCount() const
{
#if defined(_M_X86_64)
return m_jit ? m_jit->GetInstructionCacheClearCount() : 0;
#else
return 0;
#endif
}
u64 ARMCore::GetJitCodeSpaceClearCount() const
{
#if defined(_M_X86_64)
return m_jit ? m_jit->GetCodeSpaceClearCount() : 0;
#else
return 0;
#endif
}
u64 ARMCore::GetJitDiscardedBlockCount() const
{
#if defined(_M_X86_64)
return m_jit ? m_jit->GetDiscardedBlockCount() : 0;
#else
return 0;
#endif
}
u64 ARMCore::GetJitAddressTranslationCount() const
{
#if defined(_M_X86_64)
@@ -990,6 +1250,16 @@ u64 ARMCore::GetJitSlowSRAMPageWriteAccessCount(size_t page) const
#endif
}
std::vector<std::pair<u64, u64>> ARMCore::TakeHotJitSlowMemorySamples(size_t maximum_count)
{
#if defined(_M_X86_64)
return m_jit ? m_jit->TakeHotSlowMemorySamples(maximum_count) :
std::vector<std::pair<u64, u64>>{};
#else
return {};
#endif
}
void ARMCore::RecordJitFallback(u32 address, bool thumb)
{
++m_jit_fallback_instruction_count;
+34
View File
@@ -64,6 +64,10 @@ public:
virtual u8* GetFastmemSRAMBase() const { return nullptr; }
virtual const bool* GetFastmemBoot0Mapped() const { return nullptr; }
virtual const bool* GetFastmemSRAMSplitMode() const { return nullptr; }
// Returns a stable host pointer for ordinary physical memory. The pointer is used only to
// validate ARM page-table descriptors after guest TLB maintenance; MMIO and overlaid ROM must
// return nullptr so their observable reads continue through the bus.
virtual const u8* GetDirectMemoryPointer(u32 address, u32 size) const { return nullptr; }
};
class ARMCore final
@@ -164,6 +168,29 @@ public:
u64 GetJitExecutedInstructions() const;
u64 GetJitNativeExecutedInstructions() const;
size_t GetJitCompiledBlockCount() const;
u64 GetJitLifetimeCompiledBlockCount() const;
u64 GetJitBlockLookupCount() const;
u64 GetJitBlockCacheHitCount() const;
u64 GetJitDispatchSlowCount() const;
u64 GetJitDispatchStaleGenerationCount() const;
u64 GetJitDispatchPhysicalAliasCount() const;
u64 GetJitDispatchCollisionCount() const;
u64 GetJitDispatchKeyMissCount() const;
u64 GetJitDispatchPageMissCount() const;
u64 GetJitDispatchEmptyEntryCount() const;
u64 GetJitDispatchGeneratedKeyMismatchCount() const;
u64 GetJitDispatchGeneratedSetMismatchCount() const;
u64 GetJitDispatchKeyMissWithExactMatchCount() const;
u64 GetJitDispatchKeyMissWithPhysicalAliasCount() const;
u64 GetJitDispatchKeyMissWithSamePhysicalPageCount() const;
u64 GetJitDispatchKeyMissWithEmptySlotCount() const;
u64 GetJitDispatchKeyMissWithFullSetCount() const;
u64 GetJitDispatchKeyMissWithStaleTranslationCount() const;
u64 GetJitFastEntryEmptyInsertCount() const;
u64 GetJitFastEntryReplacementCount() const;
u64 GetJitInstructionCacheClearCount() const;
u64 GetJitCodeSpaceClearCount() const;
u64 GetJitDiscardedBlockCount() const;
u64 GetJitAddressTranslationCount() const;
u64 GetJitSlowReadCount() const;
u64 GetJitSlowWriteCount() const;
@@ -178,6 +205,7 @@ public:
u64 GetJitSlowSRAMPageAccessCount(size_t page) const;
u64 GetJitSlowSRAMPageReadAccessCount(size_t page) const;
u64 GetJitSlowSRAMPageWriteAccessCount(size_t page) const;
std::vector<std::pair<u64, u64>> TakeHotJitSlowMemorySamples(size_t maximum_count);
u64 GetJitFallbackInstructionCount() const { return m_jit_fallback_instruction_count; }
u64 GetMemoryPollEntryCount() const { return m_memory_poll_entry_count; }
std::vector<HotPCSample> GetHotPCSamples(size_t maximum_count);
@@ -246,6 +274,12 @@ private:
void Write32(u32 address, u32 value);
void WriteByte(u32 address, u8 value);
u32 TranslateVirtualAddress(u32 address) const;
u32 TranslateVirtualAddressForJit(u32 address, const u8** first_descriptor,
u32* first_descriptor_value, const u8** second_descriptor,
u32* second_descriptor_value) const;
u32 WalkPageTables(u32 modified_address, const u8** first_descriptor,
u32* first_descriptor_value, const u8** second_descriptor,
u32* second_descriptor_value) const;
u32 ReadPhysical32(u32 address) const;
void InvalidateTLB();
u16 FetchThumbInstruction(u32 address);
+559 -58
View File
@@ -76,6 +76,10 @@ ARMJitX64::ARMJitX64(ARMCore& core) : m_core(core)
m_executed_instructions_offset =
static_cast<s32>(reinterpret_cast<const u8*>(&m_core.m_executed_instructions) - base);
m_control_offset = static_cast<s32>(reinterpret_cast<const u8*>(&m_core.m_cp15.control) - base);
m_translation_table_base_offset = static_cast<s32>(
reinterpret_cast<const u8*>(&m_core.m_cp15.translation_table_base) - base);
m_domain_access_control_offset = static_cast<s32>(
reinterpret_cast<const u8*>(&m_core.m_cp15.domain_access_control) - base);
m_process_id_offset =
static_cast<s32>(reinterpret_cast<const u8*>(&m_core.m_cp15.process_id) - base);
m_tlb_generation_offset =
@@ -99,6 +103,18 @@ void ARMJitX64::PoisonMemory()
}
void ARMJitX64::Clear()
{
++m_instruction_cache_clear_count;
RequestClear();
}
void ARMJitX64::ClearForCodeSpace()
{
++m_code_space_clear_count;
RequestClear();
}
void ARMJitX64::RequestClear()
{
// CP15 I-cache maintenance can be reached through a fallback at the end of the currently
// executing host block. Defer destruction until its RET has brought us back to C++.
@@ -107,8 +123,15 @@ void ARMJitX64::Clear()
m_clear_pending = true;
return;
}
ClearCodeCache();
}
void ARMJitX64::ClearCodeCache()
{
m_discarded_block_count += m_blocks.size();
m_blocks.clear();
std::ranges::fill(m_fast_entries, FastEntry{});
std::ranges::fill(m_fast_entry_next_victim, 0);
SetCodePtr(m_block_code_begin, region + region_size);
}
@@ -117,13 +140,12 @@ void ARMJitX64::InvalidateTranslationContext()
// A TLB invalidation changes which physical page a virtual PC resolves to, but it does not
// invalidate the ARM instruction cache. Keep already translated physical code and force the
// generated dispatcher to resolve the next virtual PC through the current page tables. Every
// fast entry carries the ARMCore TLB generation, so the common invalidation is O(1). This is
// critical for IOS, which flushes its TLB on virtually every process switch; clearing the whole
// 65,536-entry array here previously consumed most of the host CPU. Generation zero is skipped by
// ARMCore. If the 32-bit counter eventually wraps back to one, clear ancient generation-one
// entries once to prevent an alias after the wrap.
// shared page translation carries the ARMCore TLB generation, so the common invalidation is
// O(1). This is critical for IOS, which flushes its TLB on virtually every process switch.
// Generation zero is skipped by ARMCore. If the 32-bit counter eventually wraps back to one,
// clear ancient generation-one page translations once to prevent a generation alias.
if (m_core.m_tlb_generation == 1)
std::ranges::fill(m_fast_entries, FastEntry{});
std::ranges::fill(m_fast_translations, FastTranslationEntry{});
}
u32 ARMJitX64::Run(u64 cycle_budget)
@@ -138,7 +160,7 @@ u32 ARMJitX64::Run(u64 cycle_budget)
if (m_clear_pending)
{
m_clear_pending = false;
Clear();
ClearCodeCache();
}
const u32 executed = static_cast<u32>(result);
m_executed_instructions += executed;
@@ -168,9 +190,11 @@ void ARMJitX64::GenerateDispatcher()
CMP(8, MatR(RAX), Imm8(0));
FixupBranch invalidated = J_CC(CC_NE, Jump::Near);
// Direct-mapped native block cache. The key includes CPSR.T in bit zero and uses the ARM926
// modified virtual address (MVA) for low FCSE addresses. Different IOS process identifiers can
// therefore retain independent hot entries without invalidating the cache on every c13 write.
// Four-way native block cache. The key includes CPSR.T in bit zero and uses the ARM926 modified
// virtual address (MVA) for low FCSE addresses. Different IOS process identifiers can therefore
// retain independent hot entries without invalidating the cache on every c13 write. Four ways
// are important here: IOS regularly alternates among several hot basic blocks whose MVAs used
// to evict each other millions of times in the old direct-mapped cache.
MOV(32, R(EAX), MRegister(15));
MOV(32, R(ECX), MCPSR());
SHR(32, R(ECX), Imm8(5));
@@ -187,30 +211,181 @@ void ARMJitX64::GenerateDispatcher()
SetJumpTarget(mmu_disabled);
SetJumpTarget(outside_fcse);
// Fold the FCSE PID bits down into the cache index. A plain low-bit mask makes every process
// collide because PID occupies MVA[31:25]. The full MVA remains in FastEntry::key for safety.
// Multiplicative hashing folds both FCSE PID and low instruction-address bits into the set. Each
// set occupies one 64-byte host cache line (four 16-byte FastEntry values).
MOV(32, R(EDX), R(EAX));
SHR(32, R(EDX), Imm8(1));
MOV(32, R(ECX), R(EAX));
SHR(32, R(ECX), Imm8(17));
XOR(32, R(EDX), R(ECX));
AND(32, R(EDX), Imm32(static_cast<u32>(FAST_ENTRY_COUNT - 1)));
SHL(64, R(RDX), Imm8(4));
IMUL(32, EDX, R(EDX), Imm32(0x9e3779b1U));
SHR(32, R(EDX), Imm8(16));
SHL(64, R(RDX), Imm8(6));
MOV(64, R(R11), ImmPtr(m_fast_entries.data()));
ADD(64, R(R11), R(RDX));
MOV(32, R(R8), MDisp(JIT_CORE, m_tlb_generation_offset));
CMP(32, MDisp(R11, static_cast<s32>(offsetof(FastEntry, tlb_generation))), R(R8));
FixupBranch stale_generation = J_CC(CC_NE);
// With the MMU disabled the virtual and physical identities are identical, so a key-only lookup
// is sufficient and TLB maintenance is irrelevant.
MOV(32, R(ECX), MDisp(JIT_CORE, m_control_offset));
TEST(32, R(ECX), Imm32(1));
FixupBranch mmu_fast_translation = J_CC(CC_NE, Jump::Near);
MOV(32, R(R8), R(EAX));
AND(32, R(R8), Imm32(~0x3ffU));
CMP(32, MDisp(R11, static_cast<s32>(offsetof(FastEntry, key))), R(EAX));
FixupBranch cache_miss = J_CC(CC_NE);
FixupBranch no_mmu_way0_key_miss = J_CC(CC_NE, Jump::Near);
CMP(32, MDisp(R11, static_cast<s32>(offsetof(FastEntry, physical_page))), R(R8));
FixupBranch no_mmu_hit_way0 = J_CC(CC_E, Jump::Near);
SetJumpTarget(no_mmu_way0_key_miss);
CMP(32, MDisp(R11, static_cast<s32>(sizeof(FastEntry) + offsetof(FastEntry, key))), R(EAX));
FixupBranch no_mmu_way1_key_miss = J_CC(CC_NE, Jump::Near);
CMP(32, MDisp(R11, static_cast<s32>(sizeof(FastEntry) + offsetof(FastEntry, physical_page))),
R(R8));
FixupBranch no_mmu_hit_way1 = J_CC(CC_E, Jump::Near);
SetJumpTarget(no_mmu_way1_key_miss);
CMP(32, MDisp(R11, static_cast<s32>(2 * sizeof(FastEntry) + offsetof(FastEntry, key))), R(EAX));
FixupBranch no_mmu_way2_key_miss = J_CC(CC_NE, Jump::Near);
CMP(32, MDisp(R11,
static_cast<s32>(2 * sizeof(FastEntry) + offsetof(FastEntry, physical_page))),
R(R8));
FixupBranch no_mmu_hit_way2 = J_CC(CC_E, Jump::Near);
SetJumpTarget(no_mmu_way2_key_miss);
CMP(32, MDisp(R11, static_cast<s32>(3 * sizeof(FastEntry) + offsetof(FastEntry, key))), R(EAX));
FixupBranch no_mmu_cache_miss_key = J_CC(CC_NE, Jump::Near);
CMP(32, MDisp(R11,
static_cast<s32>(3 * sizeof(FastEntry) + offsetof(FastEntry, physical_page))),
R(R8));
FixupBranch no_mmu_cache_miss_page = J_CC(CC_NE, Jump::Near);
ADD(64, R(R11), Imm8(static_cast<u8>(3 * sizeof(FastEntry))));
FixupBranch no_mmu_way_selected = J(Jump::Near);
SetJumpTarget(no_mmu_hit_way2);
ADD(64, R(R11), Imm8(static_cast<u8>(2 * sizeof(FastEntry))));
FixupBranch no_mmu_way2_selected = J(Jump::Near);
SetJumpTarget(no_mmu_hit_way1);
ADD(64, R(R11), Imm8(static_cast<u8>(sizeof(FastEntry))));
SetJumpTarget(no_mmu_hit_way0);
SetJumpTarget(no_mmu_way2_selected);
SetJumpTarget(no_mmu_way_selected);
MOV(64, R(R10), MDisp(R11, static_cast<s32>(offsetof(FastEntry, entry))));
TEST(64, R(R10), R(R10));
FixupBranch empty_entry_no_mmu = J_CC(CC_Z, Jump::Near);
JMPptr(R(R10));
SetJumpTarget(mmu_fast_translation);
// ARM926 TLB maintenance invalidates translations by page. Validate one shared 1 KiB physical
// identity per page instead of sending every decoded block through C++ after each IOS context
// switch. A true page remap still rejects the old native block below.
MOV(32, R(R9), R(EAX));
SHR(32, R(R9), Imm8(10));
MOV(32, R(R8), R(R9));
MOV(32, R(ECX), R(R9));
SHR(32, R(ECX), Imm8(12));
XOR(32, R(R8), R(ECX));
AND(32, R(R8), Imm32(static_cast<u32>(FAST_TRANSLATION_ENTRY_COUNT - 1)));
SHL(64, R(R8), Imm8(6));
MOV(64, R(R10), ImmPtr(m_fast_translations.data()));
ADD(64, R(R10), R(R8));
CMP(32, MDisp(R10, static_cast<s32>(offsetof(FastTranslationEntry, virtual_page))), R(R9));
FixupBranch stale_page = J_CC(CC_NE, Jump::Near);
MOV(32, R(R8), MDisp(JIT_CORE, m_tlb_generation_offset));
CMP(32, MDisp(R10, static_cast<s32>(offsetof(FastTranslationEntry, tlb_generation))), R(R8));
FixupBranch current_translation = J_CC(CC_E, Jump::Near);
// IOS invalidates its ARM926 TLB on almost every process switch, even when the page tables are
// unchanged. Revalidate the exact descriptors that established this translation in generated
// code. A changed TTBR or descriptor still takes the complete C++ page-table walk below, while
// unchanged mappings avoid millions of dispatcher side exits.
MOV(32, R(ECX), MDisp(JIT_CORE, m_translation_table_base_offset));
CMP(32, MDisp(R10, static_cast<s32>(offsetof(FastTranslationEntry, translation_table_base))),
R(ECX));
FixupBranch stale_translation_table = J_CC(CC_NE, Jump::Near);
MOV(64, R(R8), MDisp(R10, static_cast<s32>(offsetof(FastTranslationEntry, first_descriptor))));
TEST(64, R(R8), R(R8));
FixupBranch unavailable_first_descriptor = J_CC(CC_Z, Jump::Near);
MOV(32, R(ECX), MatR(R8));
CMP(32, R(ECX),
MDisp(R10, static_cast<s32>(offsetof(FastTranslationEntry, first_descriptor_value))));
FixupBranch changed_first_descriptor = J_CC(CC_NE, Jump::Near);
MOV(64, R(R8), MDisp(R10, static_cast<s32>(offsetof(FastTranslationEntry, second_descriptor))));
TEST(64, R(R8), R(R8));
FixupBranch no_second_descriptor = J_CC(CC_Z, Jump::Near);
MOV(32, R(ECX), MatR(R8));
CMP(32, R(ECX),
MDisp(R10, static_cast<s32>(offsetof(FastTranslationEntry, second_descriptor_value))));
FixupBranch changed_second_descriptor = J_CC(CC_NE, Jump::Near);
SetJumpTarget(no_second_descriptor);
MOV(32, R(R8), MDisp(JIT_CORE, m_tlb_generation_offset));
MOV(32, MDisp(R10, static_cast<s32>(offsetof(FastTranslationEntry, tlb_generation))), R(R8));
SetJumpTarget(current_translation);
MOV(32, R(R8), MDisp(R10, static_cast<s32>(offsetof(FastTranslationEntry, physical_page))));
// The same MVA can legitimately resolve to different physical pages in different IOS address
// spaces. Select a way only when both identities match, allowing the four physical mappings to
// coexist instead of overwriting one another on every process switch.
CMP(32, MDisp(R11, static_cast<s32>(offsetof(FastEntry, key))), R(EAX));
FixupBranch mmu_way0_key_miss = J_CC(CC_NE, Jump::Near);
CMP(32, MDisp(R11, static_cast<s32>(offsetof(FastEntry, physical_page))), R(R8));
FixupBranch mmu_hit_way0 = J_CC(CC_E, Jump::Near);
SetJumpTarget(mmu_way0_key_miss);
CMP(32, MDisp(R11, static_cast<s32>(sizeof(FastEntry) + offsetof(FastEntry, key))), R(EAX));
FixupBranch mmu_way1_key_miss = J_CC(CC_NE, Jump::Near);
CMP(32, MDisp(R11, static_cast<s32>(sizeof(FastEntry) + offsetof(FastEntry, physical_page))),
R(R8));
FixupBranch mmu_hit_way1 = J_CC(CC_E, Jump::Near);
SetJumpTarget(mmu_way1_key_miss);
CMP(32, MDisp(R11, static_cast<s32>(2 * sizeof(FastEntry) + offsetof(FastEntry, key))), R(EAX));
FixupBranch mmu_way2_key_miss = J_CC(CC_NE, Jump::Near);
CMP(32, MDisp(R11,
static_cast<s32>(2 * sizeof(FastEntry) + offsetof(FastEntry, physical_page))),
R(R8));
FixupBranch mmu_hit_way2 = J_CC(CC_E, Jump::Near);
SetJumpTarget(mmu_way2_key_miss);
CMP(32, MDisp(R11, static_cast<s32>(3 * sizeof(FastEntry) + offsetof(FastEntry, key))), R(EAX));
FixupBranch mmu_cache_miss_key = J_CC(CC_NE, Jump::Near);
CMP(32, MDisp(R11,
static_cast<s32>(3 * sizeof(FastEntry) + offsetof(FastEntry, physical_page))),
R(R8));
FixupBranch mmu_cache_miss_page = J_CC(CC_NE, Jump::Near);
ADD(64, R(R11), Imm8(static_cast<u8>(3 * sizeof(FastEntry))));
FixupBranch mmu_way_selected = J(Jump::Near);
SetJumpTarget(mmu_hit_way2);
ADD(64, R(R11), Imm8(static_cast<u8>(2 * sizeof(FastEntry))));
FixupBranch mmu_way2_selected = J(Jump::Near);
SetJumpTarget(mmu_hit_way1);
ADD(64, R(R11), Imm8(static_cast<u8>(sizeof(FastEntry))));
SetJumpTarget(mmu_hit_way0);
SetJumpTarget(mmu_way2_selected);
SetJumpTarget(mmu_way_selected);
MOV(64, R(R11), MDisp(R11, static_cast<s32>(offsetof(FastEntry, entry))));
TEST(64, R(R11), R(R11));
FixupBranch empty_entry = J_CC(CC_Z);
JMPptr(R(R11));
SetJumpTarget(stale_generation);
SetJumpTarget(cache_miss);
SetJumpTarget(stale_page);
SetJumpTarget(stale_translation_table);
SetJumpTarget(unavailable_first_descriptor);
SetJumpTarget(changed_first_descriptor);
SetJumpTarget(changed_second_descriptor);
MOV(32, R(ABI_PARAM4), R(EDX));
MOV(32, R(ABI_PARAM3), R(EAX));
MOV(32, R(ABI_PARAM2), Imm32(static_cast<u32>(DispatchReason::Translation)));
FixupBranch dispatch_reason_translation = J(Jump::Near);
SetJumpTarget(no_mmu_cache_miss_key);
SetJumpTarget(mmu_cache_miss_key);
MOV(32, R(ABI_PARAM4), R(EDX));
MOV(32, R(ABI_PARAM3), R(EAX));
MOV(32, R(ABI_PARAM2), Imm32(static_cast<u32>(DispatchReason::KeyMiss)));
FixupBranch dispatch_reason_key = J(Jump::Near);
SetJumpTarget(no_mmu_cache_miss_page);
SetJumpTarget(mmu_cache_miss_page);
MOV(32, R(ABI_PARAM4), R(EDX));
MOV(32, R(ABI_PARAM3), R(EAX));
MOV(32, R(ABI_PARAM2), Imm32(static_cast<u32>(DispatchReason::PageMiss)));
FixupBranch dispatch_reason_page = J(Jump::Near);
SetJumpTarget(empty_entry);
SetJumpTarget(empty_entry_no_mmu);
MOV(32, R(ABI_PARAM4), R(EDX));
MOV(32, R(ABI_PARAM3), R(EAX));
MOV(32, R(ABI_PARAM2), Imm32(static_cast<u32>(DispatchReason::EmptyEntry)));
SetJumpTarget(dispatch_reason_translation);
SetJumpTarget(dispatch_reason_key);
SetJumpTarget(dispatch_reason_page);
MOV(64, R(ABI_PARAM1), ImmPtr(this));
ABI_CallFunction(Dispatch);
TEST(64, R(RAX), R(RAX));
@@ -245,48 +420,154 @@ u32 ARMJitX64::MakeFastEntryKey(u32 address, u32 control, u32 process_id)
return address;
}
size_t ARMJitX64::GetFastEntryIndex(u32 key)
size_t ARMJitX64::GetFastEntrySetIndex(u32 key)
{
return ((key >> 1) ^ (key >> 17)) & (FAST_ENTRY_COUNT - 1);
return static_cast<u32>(key * 0x9e3779b1U) >> 16;
}
const u8* ARMJitX64::Dispatch(ARMJitX64* jit)
const u8* ARMJitX64::Dispatch(ARMJitX64* jit, DispatchReason reason, u32 generated_key,
u32 generated_set_offset)
{
const bool thumb = (jit->m_core.m_cpsr & ARMCore::CPSR_T) != 0;
const u32 key = jit->m_core.m_registers[15] | (thumb ? 1U : 0U);
Block* const block = jit->GetOrCompileBlock(key);
if (!block || !block->runnable)
return nullptr;
const u32 fast_key =
MakeFastEntryKey(key, jit->m_core.m_cp15.control, jit->m_core.m_cp15.process_id);
FastEntry& fast_entry = jit->m_fast_entries[GetFastEntryIndex(fast_key)];
fast_entry.key = fast_key;
fast_entry.tlb_generation = jit->m_core.m_tlb_generation;
fast_entry.entry = block->entry;
const u32 virtual_page = fast_key >> 10;
FastTranslationEntry& translation =
jit->m_fast_translations[(virtual_page ^ (virtual_page >> 12)) &
(FAST_TRANSLATION_ENTRY_COUNT - 1)];
const bool mmu_enabled = (jit->m_core.m_cp15.control & 1) != 0;
const bool translation_is_current =
!mmu_enabled || (translation.tlb_generation == jit->m_core.m_tlb_generation &&
translation.virtual_page == virtual_page);
const size_t set_index = GetFastEntrySetIndex(fast_key);
const size_t set_base = set_index * FAST_ENTRY_WAYS;
if (generated_key != fast_key)
++jit->m_dispatch_generated_key_mismatch_count;
if (generated_set_offset != set_index * FAST_ENTRY_WAYS * sizeof(FastEntry))
++jit->m_dispatch_generated_set_mismatch_count;
++jit->m_dispatch_slow_count;
if (reason == DispatchReason::Translation)
++jit->m_dispatch_stale_generation_count;
else if (reason == DispatchReason::KeyMiss)
++jit->m_dispatch_key_miss_count;
else if (reason == DispatchReason::PageMiss)
++jit->m_dispatch_page_miss_count;
else if (reason == DispatchReason::EmptyEntry)
++jit->m_dispatch_empty_entry_count;
u32 physical_address = 0;
const u8* first_descriptor = nullptr;
const u8* second_descriptor = nullptr;
u32 first_descriptor_value = 0;
u32 second_descriptor_value = 0;
Block* const block =
jit->GetOrCompileBlock(key, &physical_address, &first_descriptor, &first_descriptor_value,
&second_descriptor, &second_descriptor_value);
if (!block || !block->runnable)
return nullptr;
const u32 physical_page = physical_address & ~0x3ffU;
FastEntry* matching_entry = nullptr;
FastEntry* empty_entry = nullptr;
bool has_other_physical_mapping = false;
bool has_same_physical_page = false;
for (size_t way = 0; way < FAST_ENTRY_WAYS; ++way)
{
FastEntry& entry = jit->m_fast_entries[set_base + way];
if (entry.key == fast_key && entry.physical_page == physical_page)
{
matching_entry = &entry;
break;
}
if (entry.key == fast_key && entry.entry != nullptr)
has_other_physical_mapping = true;
if (entry.physical_page == physical_page && entry.entry != nullptr)
has_same_physical_page = true;
if (entry.entry == nullptr && empty_entry == nullptr)
empty_entry = &entry;
}
if (translation_is_current && matching_entry == nullptr)
{
if (has_other_physical_mapping)
++jit->m_dispatch_physical_alias_count;
else if (empty_entry == nullptr)
++jit->m_dispatch_collision_count;
}
if (reason == DispatchReason::KeyMiss)
{
if (matching_entry != nullptr)
++jit->m_dispatch_key_miss_with_exact_match_count;
if (has_other_physical_mapping)
++jit->m_dispatch_key_miss_with_physical_alias_count;
if (has_same_physical_page)
++jit->m_dispatch_key_miss_with_same_physical_page_count;
if (empty_entry != nullptr)
++jit->m_dispatch_key_miss_with_empty_slot_count;
else
++jit->m_dispatch_key_miss_with_full_set_count;
if (!translation_is_current)
++jit->m_dispatch_key_miss_with_stale_translation_count;
}
translation.virtual_page = virtual_page;
translation.physical_page = physical_page;
translation.tlb_generation = jit->m_core.m_tlb_generation;
translation.translation_table_base = jit->m_core.m_cp15.translation_table_base;
translation.first_descriptor = first_descriptor;
translation.second_descriptor = second_descriptor;
translation.first_descriptor_value = first_descriptor_value;
translation.second_descriptor_value = second_descriptor_value;
FastEntry* fast_entry = matching_entry;
if (fast_entry == nullptr)
{
if (empty_entry != nullptr)
{
fast_entry = empty_entry;
++jit->m_fast_entry_empty_insert_count;
}
else
{
u8& next_victim = jit->m_fast_entry_next_victim[set_index];
fast_entry = &jit->m_fast_entries[set_base + next_victim];
next_victim = static_cast<u8>((next_victim + 1) & (FAST_ENTRY_WAYS - 1));
++jit->m_fast_entry_replacement_count;
}
}
fast_entry->key = fast_key;
fast_entry->physical_page = physical_page;
fast_entry->entry = block->entry;
return block->entry;
}
ARMJitX64::Block* ARMJitX64::GetOrCompileBlock(u32 address)
ARMJitX64::Block* ARMJitX64::GetOrCompileBlock(u32 address, u32* physical_address_out,
const u8** first_descriptor_out,
u32* first_descriptor_value_out,
const u8** second_descriptor_out,
u32* second_descriptor_value_out)
{
++m_block_lookup_count;
const bool thumb = (address & 1) != 0;
const u32 virtual_address = address & ~1U;
const u32 physical_address = m_core.TranslateVirtualAddress(virtual_address);
const u32 physical_address = m_core.TranslateVirtualAddressForJit(
virtual_address, first_descriptor_out, first_descriptor_value_out, second_descriptor_out,
second_descriptor_value_out);
*physical_address_out = physical_address;
const u64 block_key = (static_cast<u64>(physical_address) << 32) | address;
if (const auto it = m_blocks.find(block_key); it != m_blocks.end())
{
++m_block_cache_hit_count;
return &it->second;
}
if (IsAlmostFull())
{
if (m_is_running)
{
m_clear_pending = true;
ClearForCodeSpace();
return nullptr;
}
Clear();
}
Block block = CompileBlock(virtual_address, thumb);
++m_lifetime_compiled_block_count;
return &m_blocks.emplace(block_key, block).first->second;
}
@@ -469,11 +750,60 @@ ARMJitX64::Block ARMJitX64::CompileBlock(u32 address, bool thumb)
return {.entry = entry,
.instruction_count = instruction_count,
.native_instruction_count = native_instruction_count,
.runnable = native_instruction_count != 0};
// A fallback-only block is still translated host code: it calls the exact interpreter
// helper, accounts the guest instruction, and returns through the native dispatcher.
// Rejecting it here made every unsupported hot instruction repeat address translation,
// unordered-map lookup and ABI dispatch in C++ before ARMCore interpreted it anyway.
// Caching the already-emitted wrapper preserves identical instruction semantics while
// removing that redundant lookup path.
.runnable = true};
}
bool ARMJitX64::EmitDirectARM(u32 instruction, u32 address, bool* terminal, bool* dispatcher_exit)
{
// The IOS scheduler executes a small set of ARM926 system-control writes in virtually every
// syscall and interrupt path. Decode those architectural operations once while compiling the
// block instead of entering the complete interpreter decoder every time the block runs.
//
// Keep TLB maintenance terminal: software may have changed a page table immediately before the
// MCR, so compiling subsequent guest instructions through the pre-invalidation mapping would be
// incorrect. The helper performs the same generation change as ARMCore::WriteCP15 and resumes at
// the following ARM instruction through the generated dispatcher.
const bool is_mcr_p15 = (instruction & 0x0f100f10) == 0x0e000f10;
if ((instruction >> 28) == 0xe && is_mcr_p15)
{
const u32 opcode1 = (instruction >> 21) & 7;
const u32 crn = (instruction >> 16) & 0xf;
const u32 rd = (instruction >> 12) & 0xf;
const u32 crm = instruction & 0xf;
const u32 opcode2 = (instruction >> 5) & 7;
if (opcode1 == 0 && crn == 3 && rd != 15)
{
MOV(32, R(EAX), MRegister(rd));
MOV(32, MDomainAccessControl(), R(EAX));
return true;
}
// ARMCore currently models only WFI and instruction-cache-affecting c7 operations. All other
// c7 writes are architecturally harmless in our single-host-thread memory model. In
// particular IOS's c7,c6,1 and c7,c10,1 forms are among its hottest privileged instructions.
const bool is_wfi = crn == 7 && crm == 0 && opcode2 == 4;
const bool affects_instruction_cache = crn == 7 && (crm == 5 || crm == 7);
if (opcode1 == 0 && crn == 7 && !is_wfi && !affects_instruction_cache)
return true;
if (opcode1 == 0 && crn == 8)
{
FlushRegisterCache();
MOV(64, R(ABI_PARAM1), ImmPtr(this));
MOV(32, R(ABI_PARAM2), Imm32(address));
ABI_CallFunction(InvalidateTLB);
*terminal = true;
return true;
}
}
// MCR p15, 0, Rd, c7, c10, 4 is ARM926 Drain Write Buffer. It is an ordering barrier, not an
// instruction-cache invalidation, and has no additional observable work in this single-host-
// thread memory model. IOS executes it in hot synchronization paths.
@@ -640,6 +970,31 @@ bool ARMJitX64::EmitDirectARM(u32 instruction, u32 address, bool* terminal, bool
return true;
}
// IOS's crypto/bootstrap code contains tight UMULL loops. A single such instruction accounted
// for more than a million interpreter fallbacks during the first seconds of a real NAND boot.
// Translate the complete ARMv5 long-multiply family here; this keeps the arithmetic and optional
// NZ update architectural while eliminating the generic decoder/dispatcher round trip.
if ((instruction & 0x0f8000f0) == 0x00800090)
{
const u32 rd_hi = (instruction >> 16) & 0xf;
const u32 rd_lo = (instruction >> 12) & 0xf;
const u32 rs = (instruction >> 8) & 0xf;
const u32 rm = instruction & 0xf;
if (condition == 0xf || rd_hi == 15 || rd_lo == 15 || rs == 15 || rm == 15)
return false;
if (condition == 0xe)
return EmitARMMultiplyLong(instruction);
EmitConditionResult(condition);
TEST(32, R(EAX), R(EAX));
const FixupBranch predicate_failed = J_CC(CC_Z, Jump::Near);
const bool emitted = EmitARMMultiplyLong(instruction);
ASSERT(emitted);
SetJumpTarget(predicate_failed);
return true;
}
if ((instruction & 0x0e000000) == 0x0a000000 && (instruction >> 28) != 0xf)
{
const bool link = (instruction & (1U << 24)) != 0;
@@ -686,7 +1041,7 @@ bool ARMJitX64::EmitDirectARM(u32 instruction, u32 address, bool* terminal, bool
EmitConditionResult(condition);
TEST(32, R(EAX), R(EAX));
const FixupBranch predicate_failed = J_CC(CC_Z, Jump::Near);
const bool emitted = EmitARMMemory(instruction, address);
const bool emitted = EmitARMMemory(instruction, address, terminal);
ASSERT(emitted);
SetJumpTarget(predicate_failed);
return true;
@@ -707,7 +1062,7 @@ bool ARMJitX64::EmitDirectARM(u32 instruction, u32 address, bool* terminal, bool
}
if ((instruction & 0x0c000000) == 0x04000000)
return EmitARMMemory(instruction, address);
return EmitARMMemory(instruction, address, terminal);
if ((instruction & 0x0e000090) == 0x00000090)
return EmitARMHalfwordMemory(instruction, address);
@@ -722,6 +1077,73 @@ bool ARMJitX64::EmitDirectARM(u32 instruction, u32 address, bool* terminal, bool
return true;
}
bool ARMJitX64::EmitARMMultiplyLong(u32 instruction)
{
if ((instruction & 0x0f8000f0) != 0x00800090)
return false;
const bool signed_multiply = (instruction & (1U << 22)) != 0;
const bool accumulate = (instruction & (1U << 21)) != 0;
const bool set_flags = (instruction & (1U << 20)) != 0;
const u32 rd_hi = (instruction >> 16) & 0xf;
const u32 rd_lo = (instruction >> 12) & 0xf;
const u32 rs = (instruction >> 8) & 0xf;
const u32 rm = instruction & 0xf;
if (rd_hi == 15 || rd_lo == 15 || rs == 15 || rm == 15)
return false;
// Read every guest operand before writing either destination so architecturally tolerated
// source/destination aliases behave exactly like ARMCore::ExecuteMultiplyLong.
FlushRegisterCache();
if (signed_multiply)
{
MOVSX(64, 32, RAX, MStoredRegister(rm));
MOVSX(64, 32, RCX, MStoredRegister(rs));
}
else
{
MOV(32, R(EAX), MStoredRegister(rm));
MOV(32, R(ECX), MStoredRegister(rs));
}
IMUL(64, RAX, R(RCX));
if (accumulate)
{
MOV(32, R(R8), MStoredRegister(rd_hi));
SHL(64, R(R8), Imm8(32));
MOV(32, R(R9), MStoredRegister(rd_lo));
OR(64, R(R8), R(R9));
ADD(64, R(RAX), R(R8));
}
if (set_flags)
{
TEST(64, R(RAX), R(RAX));
SETcc(CC_S, R(R8));
SETcc(CC_Z, R(R9));
MOVZX(32, 8, R8, R(R8));
MOVZX(32, 8, R9, R(R9));
SHL(32, R(R8), Imm8(31));
SHL(32, R(R9), Imm8(30));
}
MOV(32, MStoredRegister(rd_lo), R(EAX));
MOV(64, R(RDX), R(RAX));
SHR(64, R(RDX), Imm8(32));
MOV(32, MStoredRegister(rd_hi), R(EDX));
if (set_flags)
{
MOV(32, R(ECX), MCPSR());
AND(32, R(ECX), Imm32(~(ARMCore::CPSR_N | ARMCore::CPSR_Z)));
OR(32, R(ECX), R(R8));
OR(32, R(ECX), R(R9));
MOV(32, MCPSR(), R(ECX));
}
LoadRegisterCache();
return true;
}
bool ARMJitX64::CanEmitARMDataProcessing(u32 instruction) const
{
// The special encodings in the data-processing space stay on the exact interpreter path. The
@@ -759,24 +1181,18 @@ bool ARMJitX64::CanEmitARMDataProcessing(u32 instruction) const
const bool shifted_register_operand = !immediate && (instruction & 0xff0) != 0;
const bool shift_by_register = shifted_register_operand && (instruction & (1U << 4)) != 0;
const u32 rs = (instruction >> 8) & 0xf;
// Carry-out from a shifted operand is only architecturally visible for logical flag-setting
// operations. Keep those on the interpreter for now; all non-flag-setting ALU forms can use the
// value-only native shifter exactly.
if (shifted_register_operand && (set_flags || !writes_result))
// Register-controlled shifts have several ARM-only carry corner cases for counts >= 32. Keep
// only those flag-setting forms on the interpreter; immediate shifts can materialize their exact
// shifter carry cheaply below.
if (shift_by_register && (set_flags || !writes_result))
return false;
if (shift_by_register && rs == 15)
return false;
const bool logical = opcode == 0 || opcode == 1 || opcode == 8 || opcode == 9 || opcode == 0xc ||
opcode == 0xd || opcode == 0xe || opcode == 0xf;
const u32 rotate = ((instruction >> 8) & 0xf) * 2;
if (logical && (set_flags || !writes_result) && immediate && rotate != 0)
return false;
return true;
}
bool ARMJitX64::EmitARMMemory(u32 instruction, u32 address)
bool ARMJitX64::EmitARMMemory(u32 instruction, u32 address, bool* terminal)
{
if (!CanEmitARMMemory(instruction))
return false;
@@ -878,8 +1294,22 @@ bool ARMJitX64::EmitARMMemory(u32 instruction, u32 address)
SHL(32, R(ECX), Imm8(3));
ROR(32, R(EAX), R(ECX));
}
if (rd == 15)
{
// ARMv5 LDR PC is an interworking branch. Preserve bit zero as the new Thumb state and
// branch to the aligned target. This is the exact operation used by Starlet's high IRQ and
// reset vectors, and keeping it in native code avoids a generic decoder call per interrupt.
MOV(32, R(ABI_PARAM3), R(EAX));
MOV(32, R(ABI_PARAM2), Imm32(address));
MOV(64, R(ABI_PARAM1), ImmPtr(this));
ABI_CallFunction(LoadPC);
*terminal = true;
}
else
{
MOV(32, MRegister(rd), R(EAX));
}
}
else
{
MOV(32, R(EDX), MRegister(rd));
@@ -907,7 +1337,10 @@ bool ARMJitX64::CanEmitARMMemory(u32 instruction) const
const u32 rd = (instruction >> 12) & 0xf;
const bool register_offset = (instruction & (1U << 25)) != 0;
const u32 rm = instruction & 0xf;
return rd != 15 && !(rn == 15 && (!preindex || writeback)) && !(load && writeback && rn == rd) &&
const bool direct_pc_load = load && rd == 15 && (instruction >> 28) == 0xe && preindex &&
!writeback && (instruction & (1U << 22)) == 0;
return (rd != 15 || direct_pc_load) && !(rn == 15 && (!preindex || writeback)) &&
!(load && writeback && rn == rd) &&
!(register_offset && (instruction & (1U << 4)) != 0) && !(register_offset && rm == 15);
}
@@ -1123,11 +1556,21 @@ void ARMJitX64::EmitARMDataProcessing(u32 instruction)
const u32 rd = (instruction >> 12) & 0xf;
const u32 rm = instruction & 0xf;
const bool writes_result = opcode < 8 || opcode > 0xb;
const bool logical = opcode == 0 || opcode == 1 || opcode == 8 || opcode == 9 || opcode == 0xc ||
opcode == 0xd || opcode == 0xe || opcode == 0xf;
const bool logical_flags = logical && (set_flags || !writes_result);
bool logical_carry_known = false;
if (immediate)
{
const u32 rotate = ((instruction >> 8) & 0xf) * 2;
MOV(32, R(EDX), Imm32(std::rotr(instruction & 0xff, rotate)));
const u32 operand = std::rotr(instruction & 0xff, rotate);
MOV(32, R(EDX), Imm32(operand));
if (logical_flags && rotate != 0)
{
MOV(32, R(R10), Imm32(operand >> 31));
logical_carry_known = true;
}
}
else
{
@@ -1168,6 +1611,21 @@ void ARMJitX64::EmitARMDataProcessing(u32 instruction)
else
{
const u32 amount = (instruction >> 7) & 0x1f;
if (logical_flags)
{
// ARM's logical S forms take C from the barrel shifter. Save that source bit before EDX
// is shifted; rotate zero preserves the old C only for the unshifted LSL #0 encoding,
// which never enters this branch.
MOV(32, R(R10), R(EDX));
if (shift_type == 0)
SHR(32, R(R10), Imm8(32 - amount));
else if (shift_type == 1 || shift_type == 2)
SHR(32, R(R10), Imm8(amount == 0 ? 31 : amount - 1));
else if (amount != 0)
SHR(32, R(R10), Imm8(amount - 1));
AND(32, R(R10), Imm8(1));
logical_carry_known = true;
}
if (shift_type == 0)
{
if (amount != 0)
@@ -1253,6 +1711,8 @@ void ARMJitX64::EmitARMDataProcessing(u32 instruction)
{
if (arithmetic)
EmitArithmeticFlags(opcode == 2 || opcode == 3 || opcode == 0xa);
else if (logical_carry_known)
EmitLogicalFlagsWithCarry(EAX, R10);
else
EmitLogicalFlags(EAX);
}
@@ -1869,6 +2329,8 @@ void ARMJitX64::EmitFastmemAddress(std::vector<FixupBranch>* slow_paths, u32 acc
{
CMP(32, R(EDX), Imm32(0x00));
const FixupBranch profiled_read_page_00 = J_CC(CC_E, Jump::Near);
CMP(32, R(EDX), Imm32(0x10));
const FixupBranch irq_vector_read_page_10 = J_CC(CC_E, Jump::Near);
CMP(32, R(EDX), Imm32(0x12));
const FixupBranch profiled_read_page_12 = J_CC(CC_E, Jump::Near);
CMP(32, R(EDX), Imm32(0x14));
@@ -1878,6 +2340,7 @@ void ARMJitX64::EmitFastmemAddress(std::vector<FixupBranch>* slow_paths, u32 acc
CMP(32, R(EDX), Imm32(0x1e));
slow_paths->push_back(J_CC(CC_NE, Jump::Near));
SetJumpTarget(profiled_read_page_00);
SetJumpTarget(irq_vector_read_page_10);
SetJumpTarget(profiled_read_page_12);
SetJumpTarget(profiled_read_page_14);
SetJumpTarget(profiled_read_page_19);
@@ -2368,6 +2831,11 @@ OpArg ARMJitX64::MExecutedInstructions() const
return MDisp(R15, m_executed_instructions_offset);
}
OpArg ARMJitX64::MDomainAccessControl() const
{
return MDisp(R15, m_domain_access_control_offset);
}
void ARMJitX64::FallbackThumb(ARMJitX64* jit, u16 instruction, u32 address)
{
ARMCore* const core = &jit->m_core;
@@ -2417,6 +2885,23 @@ void ARMJitX64::WritePSR(ARMJitX64* jit, u32 spsr, u32 field_mask, u32 value)
jit->m_core.WritePSR(spsr != 0, field_mask, value);
}
void ARMJitX64::InvalidateTLB(ARMJitX64* jit, u32 address)
{
ARMCore& core = jit->m_core;
core.m_instruction_address = address;
core.m_pc_written = false;
core.InvalidateTLB();
core.m_registers[15] = address + 4;
}
void ARMJitX64::LoadPC(ARMJitX64* jit, u32 address, u32 target)
{
ARMCore& core = jit->m_core;
core.m_instruction_address = address;
core.m_pc_written = false;
core.WritePC(target, true);
}
void ARMJitX64::ExecuteUserBankBlockTransfer(ARMJitX64* jit, u32 instruction, u32 address)
{
ARMCore& core = jit->m_core;
@@ -2444,6 +2929,8 @@ u32 ARMJitX64::ReadMemorySlow(ARMJitX64* jit, u32 physical_address, u32 access_s
u32 byte_offset)
{
++jit->m_slow_read_count;
if ((jit->m_slow_read_count & 0xff) == 0)
++jit->m_slow_memory_address_samples[physical_address & ~3ULL];
switch (ClassifySlowMemoryAddress(physical_address))
{
case SlowMemoryRegion::RAM:
@@ -2475,6 +2962,8 @@ u32 ARMJitX64::ReadMemorySlow(ARMJitX64* jit, u32 physical_address, u32 access_s
void ARMJitX64::WriteMemorySlow(ARMJitX64* jit, u32 physical_address, u32 access_size, u32 value)
{
++jit->m_slow_write_count;
if ((jit->m_slow_write_count & 0xff) == 0)
++jit->m_slow_memory_address_samples[(1ULL << 32) | (physical_address & ~3ULL)];
switch (ClassifySlowMemoryAddress(physical_address))
{
case SlowMemoryRegion::RAM:
@@ -2505,6 +2994,18 @@ void ARMJitX64::WriteMemorySlow(ARMJitX64* jit, u32 physical_address, u32 access
core.m_bus.Write32(physical_address & ~3U, value);
}
std::vector<std::pair<u64, u64>> ARMJitX64::TakeHotSlowMemorySamples(size_t maximum_count)
{
std::vector<std::pair<u64, u64>> sorted(m_slow_memory_address_samples.begin(),
m_slow_memory_address_samples.end());
m_slow_memory_address_samples.clear();
std::ranges::sort(sorted, {}, [](const auto& entry) { return entry.second; });
if (sorted.size() > maximum_count)
sorted.erase(sorted.begin(), sorted.end() - maximum_count);
std::ranges::reverse(sorted);
return sorted;
}
u32 ARMJitX64::TranslateAddress(ARMJitX64* jit, u32 address)
{
++jit->m_address_translation_count;
+123 -7
View File
@@ -35,6 +35,53 @@ public:
u64 GetExecutedInstructions() const { return m_executed_instructions; }
u64 GetNativeExecutedInstructions() const { return m_native_executed_instructions; }
size_t GetCompiledBlockCount() const { return m_blocks.size(); }
u64 GetLifetimeCompiledBlockCount() const { return m_lifetime_compiled_block_count; }
u64 GetBlockLookupCount() const { return m_block_lookup_count; }
u64 GetBlockCacheHitCount() const { return m_block_cache_hit_count; }
u64 GetDispatchSlowCount() const { return m_dispatch_slow_count; }
u64 GetDispatchStaleGenerationCount() const { return m_dispatch_stale_generation_count; }
u64 GetDispatchPhysicalAliasCount() const { return m_dispatch_physical_alias_count; }
u64 GetDispatchCollisionCount() const { return m_dispatch_collision_count; }
u64 GetDispatchKeyMissCount() const { return m_dispatch_key_miss_count; }
u64 GetDispatchPageMissCount() const { return m_dispatch_page_miss_count; }
u64 GetDispatchEmptyEntryCount() const { return m_dispatch_empty_entry_count; }
u64 GetDispatchGeneratedKeyMismatchCount() const
{
return m_dispatch_generated_key_mismatch_count;
}
u64 GetDispatchGeneratedSetMismatchCount() const
{
return m_dispatch_generated_set_mismatch_count;
}
u64 GetDispatchKeyMissWithExactMatchCount() const
{
return m_dispatch_key_miss_with_exact_match_count;
}
u64 GetDispatchKeyMissWithPhysicalAliasCount() const
{
return m_dispatch_key_miss_with_physical_alias_count;
}
u64 GetDispatchKeyMissWithSamePhysicalPageCount() const
{
return m_dispatch_key_miss_with_same_physical_page_count;
}
u64 GetDispatchKeyMissWithEmptySlotCount() const
{
return m_dispatch_key_miss_with_empty_slot_count;
}
u64 GetDispatchKeyMissWithFullSetCount() const
{
return m_dispatch_key_miss_with_full_set_count;
}
u64 GetDispatchKeyMissWithStaleTranslationCount() const
{
return m_dispatch_key_miss_with_stale_translation_count;
}
u64 GetFastEntryEmptyInsertCount() const { return m_fast_entry_empty_insert_count; }
u64 GetFastEntryReplacementCount() const { return m_fast_entry_replacement_count; }
u64 GetInstructionCacheClearCount() const { return m_instruction_cache_clear_count; }
u64 GetCodeSpaceClearCount() const { return m_code_space_clear_count; }
u64 GetDiscardedBlockCount() const { return m_discarded_block_count; }
u64 GetAddressTranslationCount() const { return m_address_translation_count; }
u64 GetSlowReadCount() const { return m_slow_read_count; }
u64 GetSlowWriteCount() const { return m_slow_write_count; }
@@ -62,6 +109,7 @@ public:
m_slow_sram_page_write_access_count[page] :
0;
}
std::vector<std::pair<u64, u64>> TakeHotSlowMemorySamples(size_t maximum_count);
private:
using RunEntry = u64 (*)(u32);
@@ -84,21 +132,54 @@ private:
struct FastEntry
{
u32 key = 0xffffffff;
u32 tlb_generation = 0;
u32 physical_page = 0xffffffff;
const u8* entry = nullptr;
};
static_assert(sizeof(FastEntry) == 16);
// TLB maintenance invalidates translations, not every decoded instruction on a page. Keep one
// generation-tagged physical identity per 1 KiB ARM926 TLB granule so the first block after a
// flush performs the page-table walk and the remaining blocks can safely retain native code.
struct alignas(64) FastTranslationEntry
{
u32 virtual_page = 0xffffffff;
u32 physical_page = 0xffffffff;
u32 tlb_generation = 0;
u32 translation_table_base = 0xffffffff;
const u8* first_descriptor = nullptr;
const u8* second_descriptor = nullptr;
u32 first_descriptor_value = 0;
u32 second_descriptor_value = 0;
std::array<u32, 6> padding{};
};
static_assert(sizeof(FastTranslationEntry) == 64);
enum class DispatchReason : u32
{
Translation,
KeyMiss,
PageMiss,
EmptyEntry,
};
void PoisonMemory() override;
void RequestClear();
void ClearCodeCache();
void ClearForCodeSpace();
void GenerateDispatcher();
static u32 MakeFastEntryKey(u32 address, u32 control, u32 process_id);
static size_t GetFastEntryIndex(u32 key);
static const u8* Dispatch(ARMJitX64* jit);
Block* GetOrCompileBlock(u32 address);
static size_t GetFastEntrySetIndex(u32 key);
static const u8* Dispatch(ARMJitX64* jit, DispatchReason reason, u32 generated_key,
u32 generated_set_offset);
Block* GetOrCompileBlock(u32 address, u32* physical_address_out,
const u8** first_descriptor_out, u32* first_descriptor_value_out,
const u8** second_descriptor_out, u32* second_descriptor_value_out);
Block CompileBlock(u32 address, bool thumb);
bool EmitDirectARM(u32 instruction, u32 address, bool* terminal, bool* dispatcher_exit);
bool EmitARMMultiplyLong(u32 instruction);
bool CanEmitARMDataProcessing(u32 instruction) const;
bool CanEmitARMMemory(u32 instruction) const;
bool EmitARMMemory(u32 instruction, u32 address);
bool EmitARMMemory(u32 instruction, u32 address, bool* terminal);
bool EmitARMHalfwordMemory(u32 instruction, u32 address);
bool EmitARMBlockTransfer(u32 instruction, u32 address, bool* terminal);
void EmitARMDataProcessing(u32 instruction);
@@ -137,11 +218,14 @@ private:
Gen::OpArg MWaitingForMemoryPoll() const;
Gen::OpArg MYieldRequested() const;
Gen::OpArg MExecutedInstructions() const;
Gen::OpArg MDomainAccessControl() const;
static void FallbackThumb(ARMJitX64* jit, u16 instruction, u32 address);
static void FallbackARM(ARMJitX64* jit, u32 instruction, u32 address);
static u32 ReadSPSR(ARMJitX64* jit);
static void WritePSR(ARMJitX64* jit, u32 spsr, u32 field_mask, u32 value);
static void InvalidateTLB(ARMJitX64* jit, u32 address);
static void LoadPC(ARMJitX64* jit, u32 address, u32 target);
static void ExecuteUserBankBlockTransfer(ARMJitX64* jit, u32 instruction, u32 address);
static void ExecuteThumbPushPop(ARMJitX64* jit, u16 instruction, u32 address);
static void ExceptionReturn(ARMJitX64* jit, u32 target);
@@ -154,11 +238,17 @@ private:
static constexpr size_t CODE_SIZE = 32 * 1024 * 1024;
static constexpr u32 MAX_BLOCK_INSTRUCTIONS = 32;
static constexpr size_t FAST_ENTRY_COUNT = 1 << 16;
static constexpr size_t FAST_ENTRY_SET_COUNT = 1 << 16;
static constexpr size_t FAST_ENTRY_WAYS = 4;
static constexpr size_t FAST_ENTRY_COUNT = FAST_ENTRY_SET_COUNT * FAST_ENTRY_WAYS;
static_assert((FAST_ENTRY_WAYS & (FAST_ENTRY_WAYS - 1)) == 0);
static constexpr size_t FAST_TRANSLATION_ENTRY_COUNT = 1 << 12;
ARMCore& m_core;
std::unordered_map<u64, Block> m_blocks;
std::array<FastEntry, FAST_ENTRY_COUNT> m_fast_entries{};
alignas(64) std::array<FastEntry, FAST_ENTRY_COUNT> m_fast_entries{};
std::array<u8, FAST_ENTRY_SET_COUNT> m_fast_entry_next_victim{};
std::array<FastTranslationEntry, FAST_TRANSLATION_ENTRY_COUNT> m_fast_translations{};
u8* m_fastmem_base = nullptr;
u8* m_sram_base = nullptr;
const bool* m_boot0_mapped = nullptr;
@@ -169,6 +259,29 @@ private:
u8* m_block_code_begin = nullptr;
u64 m_executed_instructions = 0;
u64 m_native_executed_instructions = 0;
u64 m_lifetime_compiled_block_count = 0;
u64 m_block_lookup_count = 0;
u64 m_block_cache_hit_count = 0;
u64 m_dispatch_slow_count = 0;
u64 m_dispatch_stale_generation_count = 0;
u64 m_dispatch_physical_alias_count = 0;
u64 m_dispatch_collision_count = 0;
u64 m_dispatch_key_miss_count = 0;
u64 m_dispatch_page_miss_count = 0;
u64 m_dispatch_empty_entry_count = 0;
u64 m_dispatch_generated_key_mismatch_count = 0;
u64 m_dispatch_generated_set_mismatch_count = 0;
u64 m_dispatch_key_miss_with_exact_match_count = 0;
u64 m_dispatch_key_miss_with_physical_alias_count = 0;
u64 m_dispatch_key_miss_with_same_physical_page_count = 0;
u64 m_dispatch_key_miss_with_empty_slot_count = 0;
u64 m_dispatch_key_miss_with_full_set_count = 0;
u64 m_dispatch_key_miss_with_stale_translation_count = 0;
u64 m_fast_entry_empty_insert_count = 0;
u64 m_fast_entry_replacement_count = 0;
u64 m_instruction_cache_clear_count = 0;
u64 m_code_space_clear_count = 0;
u64 m_discarded_block_count = 0;
u64 m_address_translation_count = 0;
u64 m_slow_read_count = 0;
u64 m_slow_write_count = 0;
@@ -184,6 +297,7 @@ private:
u64 m_slow_sram_high_access_count = 0;
std::array<u64, 32> m_slow_sram_page_read_access_count{};
std::array<u64, 32> m_slow_sram_page_write_access_count{};
std::unordered_map<u64, u64> m_slow_memory_address_samples;
bool m_is_running = false;
bool m_clear_pending = false;
s32 m_registers_offset = 0;
@@ -195,6 +309,8 @@ private:
s32 m_yield_requested_offset = 0;
s32 m_executed_instructions_offset = 0;
s32 m_control_offset = 0;
s32 m_translation_table_base_offset = 0;
s32 m_domain_access_control_offset = 0;
s32 m_process_id_offset = 0;
s32 m_tlb_generation_offset = 0;
u32 m_compile_instruction_count = 0;
+32
View File
@@ -296,6 +296,32 @@ void Starlet::RunSlice(s64 cycles_late)
m_core->GetCPSR(), m_system.GetWiiIPC().ReadStarletRegister(0x38),
m_system.GetWiiIPC().ReadStarletRegister(0x3c),
m_system.GetWiiIPC().ReadStarletRegister(0x40));
INFO_LOG_FMT(IOS,
"Starlet JIT cache: live-blocks={} lifetime-compiles={} lookups={} hits={} "
"slow-dispatches={} stale-generations={} physical-aliases={} collisions={} "
"key-misses={} page-misses={} empty-entries={} "
"generated-key-mismatches={} generated-set-mismatches={} exact-key-misses={} "
"key-aliases={} key-same-pages={} key-empty-slots={} key-full-sets={} "
"key-stale-translations={} empty-inserts={} replacements={} "
"icache-clears={} code-space-clears={} discarded-blocks={}",
m_core->GetJitCompiledBlockCount(), m_core->GetJitLifetimeCompiledBlockCount(),
m_core->GetJitBlockLookupCount(), m_core->GetJitBlockCacheHitCount(),
m_core->GetJitDispatchSlowCount(), m_core->GetJitDispatchStaleGenerationCount(),
m_core->GetJitDispatchPhysicalAliasCount(), m_core->GetJitDispatchCollisionCount(),
m_core->GetJitDispatchKeyMissCount(), m_core->GetJitDispatchPageMissCount(),
m_core->GetJitDispatchEmptyEntryCount(),
m_core->GetJitDispatchGeneratedKeyMismatchCount(),
m_core->GetJitDispatchGeneratedSetMismatchCount(),
m_core->GetJitDispatchKeyMissWithExactMatchCount(),
m_core->GetJitDispatchKeyMissWithPhysicalAliasCount(),
m_core->GetJitDispatchKeyMissWithSamePhysicalPageCount(),
m_core->GetJitDispatchKeyMissWithEmptySlotCount(),
m_core->GetJitDispatchKeyMissWithFullSetCount(),
m_core->GetJitDispatchKeyMissWithStaleTranslationCount(),
m_core->GetJitFastEntryEmptyInsertCount(),
m_core->GetJitFastEntryReplacementCount(),
m_core->GetJitInstructionCacheClearCount(), m_core->GetJitCodeSpaceClearCount(),
m_core->GetJitDiscardedBlockCount());
std::array<u32, 4> hot_sram_read_pages{};
std::array<u64, 4> hot_sram_read_page_counts{};
std::array<u32, 4> hot_sram_write_pages{};
@@ -335,6 +361,12 @@ void Starlet::RunSlice(s64 cycles_late)
hot_sram_write_page_counts[1], hot_sram_write_pages[2],
hot_sram_write_page_counts[2], hot_sram_write_pages[3],
hot_sram_write_page_counts[3]);
for (const auto& [key, samples] : m_core->TakeHotJitSlowMemorySamples(8))
{
const bool write = (key >> 32) != 0;
INFO_LOG_FMT(IOS, "Starlet slow memory {} address={:#010x} samples={}",
write ? "write" : "read", static_cast<u32>(key), samples);
}
const ARMCore::HotPCSample current = m_core->GetCurrentPCSample();
INFO_LOG_FMT(IOS, "Starlet current PC {:#010x} {} instruction={:#010x}", current.address,
current.thumb ? "Thumb" : "ARM", current.instruction);
+77 -22
View File
@@ -189,13 +189,6 @@ constexpr u32 OHCI_PORT_CHANGE_MASK = 0x001f0000;
constexpr u32 OHCI_FRAME_CYCLES = 243000;
constexpr u16 OHCI1_ATTACH_DELAY_FRAMES = 100;
constexpr u64 WIIMOTE_UPDATE_CYCLES = 243000000 / Wiimote::UPDATE_FREQ;
// Poll host controls at Dolphin's normal 200 Hz, but do not wake the
// interpreted IOS Bluetooth stack for every poll. A 30 Hz steady HID stream
// leaves substantially more host time for Broadway; button transitions bypass
// this throttle below so presses and releases still reach IOS promptly.
constexpr u32 WIIMOTE_REPORT_FREQUENCY = 30;
constexpr u64 WIIMOTE_REPORT_CYCLES = 243000000 / WIIMOTE_REPORT_FREQUENCY;
static_assert(WIIMOTE_REPORT_FREQUENCY <= Wiimote::UPDATE_FREQ);
// Hollywood completes the internal OHCI1 port reset before IOS's first 2 ms
// poll. Using the generic 10 ms upper-bound timing leaves IOS with RHSC masked
// when PRSC arrives.
@@ -334,6 +327,7 @@ constexpr u32 HW_USBFRCRST = HW_BASE + 0x88;
constexpr u32 HW_SRNPROT = HW_BASE + 0x60;
constexpr u32 HW_AHBPROT = HW_BASE + 0x64;
constexpr u32 HW_TIMER = HW_BASE + 0x10;
constexpr u64 TIMER_CLOCK_DIVISOR = 128;
constexpr u32 HW_ALARM = HW_BASE + 0x14;
constexpr u32 HW_GPIO_ENABLE = HW_BASE + 0xdc;
constexpr u32 HW_GPIO_OUT = HW_BASE + 0xe0;
@@ -683,6 +677,7 @@ void StarletMemory::Reset()
for (size_t controller = 0; controller < OHCI_BASES.size(); ++controller)
ResetOHCIController(controller);
m_arm_cycles = 0;
m_timer_offset = 0;
m_boot0_mapped = true;
m_sram_split_mode = false;
// Retail Hollywood production revision. Early firmware branches on this
@@ -786,6 +781,7 @@ void StarletMemory::DoState(PointerWrap& p)
p.Do(m_seeprom_write_pending);
p.Do(m_seeprom_write_all);
p.Do(m_arm_cycles);
p.Do(m_timer_offset);
p.Do(m_initialized);
p.Do(m_boot0_mapped);
p.Do(m_sram_split_mode);
@@ -799,10 +795,11 @@ void StarletMemory::DoState(PointerWrap& p)
u32 StarletMemory::GetTimer() const
{
// Hollywood's 19.2 MHz timer is clocked at 32/405 of the 243 MHz Starlet
// clock.
const u64 timer = (m_arm_cycles / 405) * 32 + ((m_arm_cycles % 405) * 32) / 405;
return static_cast<u32>(timer);
// Hollywood's timer runs at Starlet / 128 (1.8984375 MHz at 243 MHz).
// Keep the divider phase across scheduler slices. The writable counter has
// its own offset so IOS resetting HW_TIMER cannot rewind peripheral time.
// https://wiibrew.org/wiki/Hardware/Starlet_Timer
return static_cast<u32>(m_arm_cycles / TIMER_CLOCK_DIVISOR) + m_timer_offset;
}
std::optional<u32> StarletMemory::TryReadBroadwayResetInstruction(u32 address) const
@@ -895,6 +892,32 @@ const bool* StarletMemory::GetFastmemSRAMSplitMode() const
return &m_sram_split_mode;
}
const u8* StarletMemory::GetDirectMemoryPointer(u32 address, u32 size) const
{
if (size == 0 || address > std::numeric_limits<u32>::max() - (size - 1))
return nullptr;
const u32 last_address = address + size - 1;
if (IsMemoryAddress(address) && IsMemoryAddress(last_address))
{
if (u8* const base = m_system.GetMemory().GetPhysicalBase())
return base + address;
}
// Page tables can live in Hollywood SRAM. Respect the same boot0 overlay and A/B split mapping
// as normal bus reads, and only expose a pointer when the whole descriptor is contiguous.
if (IsSRAMWindowAddress(address) && IsSRAMWindowAddress(last_address) &&
!IsBootROMAddress(address) && !IsBootROMAddress(last_address))
{
const u32 offset = GetSRAMOffset(address);
const u32 end_offset = GetSRAMOffset(last_address);
if (offset != INVALID_SRAM_OFFSET && end_offset == offset + size - 1)
return m_sram.data() + offset;
}
return nullptr;
}
bool StarletMemory::IsMemoryAddress(u32 address)
{
return address < Memory::MEM1_SIZE_RETAIL ||
@@ -1677,25 +1700,18 @@ void StarletMemory::UpdateWiimotes()
next_calls[i] = m_wiimotes[i]->PrepareInput(&states[i]);
}
const u64 previous_update_cycles =
m_arm_cycles >= WIIMOTE_UPDATE_CYCLES ? m_arm_cycles - WIIMOTE_UPDATE_CYCLES : 0;
const bool report_due =
m_arm_cycles / WIIMOTE_REPORT_CYCLES != previous_update_cycles / WIIMOTE_REPORT_CYCLES;
// Deliver input at the normal Wii Remote cadence, even when buttons have not
// changed. KPAD repeat timing depends on fresh reports; starving KPADRead can
// also leave a previous trigger visible across successive menu frames.
for (size_t i = 0; i < m_wiimotes.size(); ++i)
{
if (!m_wiimotes[i])
continue;
const bool button_changed = states[i].buttons.hex != m_last_wiimote_buttons[i];
if (next_calls[i] != IOS::HLE::WiimoteDevice::NextUpdateInputCall::Update || report_due ||
button_changed)
{
m_wiimotes[i]->UpdateInput(next_calls[i], states[i]);
if (next_calls[i] == IOS::HLE::WiimoteDevice::NextUpdateInputCall::Update)
m_last_wiimote_buttons[i] = states[i].buttons.hex;
}
}
}
void StarletMemory::SendACLPacket(const bdaddr_t& source, const u8* data, u32 size)
@@ -3326,6 +3342,22 @@ u16 StarletMemory::Read16(u32 address)
u32 StarletMemory::Read32(u32 address)
{
// ARM IOS accesses the Hollywood timer and interrupt controller almost exclusively as aligned
// words. Falling back to ARMBus::Read32 decomposes each access into four virtual Read8 calls;
// every byte then repeats the complete device-range decoder and IPC register switch. Preserve
// exactly the same live values while resolving these two hottest register families once.
if ((address & 3) == 0)
{
if (address == HW_TIMER)
return GetTimer();
if ((address >= HW_BASE && address <= HW_BASE + 0x0c) ||
(address >= HW_BASE + 0x30 && address <= HW_BASE + 0x40) || address == HW_AHBPROT ||
address == HW_RESETS)
{
return m_system.GetWiiIPC().ReadStarletRegister(address - HW_BASE);
}
}
if ((address & 3) == 0 && IsDIAddress(address))
{
MMIO::Mapping* const mmio = m_system.GetMemory().GetMMIOMapping();
@@ -3513,7 +3545,7 @@ void StarletMemory::Write8(u32 address, u8 value)
else if (word_address == HW_OTPCMD)
HandleOTPCommand(word);
else if (word_address == HW_TIMER)
m_arm_cycles = static_cast<u64>(word) * 405 / 32;
m_timer_offset = word - static_cast<u32>(m_arm_cycles / TIMER_CLOCK_DIVISOR);
else if (word_address == HW_ALARM)
{
// HW_ALARM is a comparator, not an interrupt acknowledgement register.
@@ -3602,6 +3634,29 @@ void StarletMemory::Write16(u32 address, u16 value)
void StarletMemory::Write32(u32 address, u32 value)
{
// Match the aligned read fast path above. WriteRegister keeps the byte-addressable backing image
// coherent, while the exact device handlers retain W1C interrupt flags, reset side effects and
// the Starlet/Broadway scheduling boundary of the four-byte Write8 path.
if ((address & 3) == 0 && address == HW_TIMER)
{
WriteRegister(address, value);
m_timer_offset = value - static_cast<u32>(m_arm_cycles / TIMER_CLOCK_DIVISOR);
return;
}
if ((address & 3) == 0 && ((address >= HW_BASE && address <= HW_BASE + 0x0c) ||
(address >= HW_BASE + 0x30 && address <= HW_BASE + 0x40) ||
address == HW_AHBPROT || address == HW_RESETS))
{
WriteRegister(address, value);
m_system.GetWiiIPC().WriteStarletRegister(address - HW_BASE, value);
if (address == HW_BASE + 0x0c && (value & 0x09) != 0)
{
if (Starlet* const starlet = m_system.GetStarlet())
starlet->YieldForIPC();
}
return;
}
if ((address & 3) == 0 && IsDIAddress(address))
{
MMIO::Mapping* const mmio = m_system.GetMemory().GetMMIOMapping();
@@ -69,6 +69,7 @@ public:
u8* GetFastmemSRAMBase() const override;
const bool* GetFastmemBoot0Mapped() const override;
const bool* GetFastmemSRAMSplitMode() const override;
const u8* GetDirectMemoryPointer(u32 address, u32 size) const override;
u64 GetCycles() const { return m_arm_cycles; }
std::optional<u32> TryReadBroadwayResetInstruction(u32 address) const;
@@ -302,6 +303,7 @@ private:
bool m_seeprom_write_pending = false;
bool m_seeprom_write_all = false;
u64 m_arm_cycles = 0;
u32 m_timer_offset = 0;
bool m_initialized = false;
bool m_boot0_mapped = true;
bool m_sram_split_mode = false;
+136
View File
@@ -3,7 +3,11 @@
#include "Core/PowerPC/Jit64/Jit.h"
#include <algorithm>
#include <chrono>
#include <cstdlib>
#include <map>
#include <optional>
#include <span>
#include <sstream>
#include <string>
@@ -18,6 +22,7 @@
#endif
#include "Common/CommonTypes.h"
#include "Common/FileUtil.h"
#include "Common/GekkoDisassembler.h"
#include "Common/HostDisassembler.h"
#include "Common/IOFile.h"
@@ -51,6 +56,111 @@
using namespace Gen;
using namespace PowerPC;
namespace
{
// Temporary, opt-in event diagnostics. Addresses and expected instructions come from a local
// tracepoint file, not from a title-specific emulation rule. No guest state is modified.
struct PPCEventTrace
{
std::map<u32, u32> points;
File::IOFile output;
std::chrono::steady_clock::time_point start;
std::chrono::steady_clock::time_point last_flush;
u32 rows = 0;
void Init()
{
points.clear();
rows = 0;
const char* config_path = std::getenv("DOLPHIN_PPC_EVENT_TRACE");
const char* output_path = std::getenv("DOLPHIN_PPC_EVENT_TRACE_OUTPUT");
if (!config_path || !output_path)
return;
std::string config;
if (!File::ReadFileToString(config_path, config))
return;
std::istringstream input(config);
u32 pc, instruction;
while (input >> std::hex >> pc >> instruction)
{
if ((pc & 3) || points.size() >= 32)
{
points.clear();
return;
}
points.emplace(pc, instruction);
}
if (points.empty() || !output.Open(output_path, "ab", File::SharedAccess::Read))
{
points.clear();
return;
}
output.WriteString("seq,wall_us,ticks,pc,lr,ctr,r3,r4,r5,r6,r7,r8,"
"mem0,mem1,mem2,mem3,mem4,mem5,mem6\n");
output.Flush();
start = last_flush = std::chrono::steady_clock::now();
}
};
PPCEventTrace s_ppc_event_trace;
// Inspect only the standard cached MEM1/MEM2 aliases. In particular, don't use an MMU load here:
// it could populate a cache line or change PLRU just by observing the packet.
std::optional<u32> PeekEventWord(Core::System& system, u32 address)
{
if ((address & 3) || (address >> 28 != 8 && address >> 28 != 9))
return std::nullopt;
const u32 physical = address & 0x3fffffff;
auto& memory = system.GetMemory();
if (!(physical < memory.GetRamSizeReal() && memory.GetRamSizeReal() - physical >= 4) &&
!(physical >= 0x10000000 && physical - 0x10000000 < memory.GetExRamSizeReal() &&
memory.GetExRamSizeReal() - (physical - 0x10000000) >= 4))
return std::nullopt;
const auto& ppc = system.GetPowerPC().GetPPCState();
const UReg_HID0 hid0{.Hex = ppc.spr[SPR_HID0]};
if (ppc.m_enable_dcache && hid0.DCE)
{
const u32 set = (physical >> 5) & 127;
const auto& cache = ppc.dCache;
for (u32 way = 0; way < 8; ++way)
{
if ((cache.valid[set] & (1 << way)) && cache.addrs[set][way] == (physical & ~31U))
return Common::swap32(cache.data[set][way][(physical & 31) / 4]);
}
}
const u8* data = memory.GetPointerForRange(physical, 4);
return data ? std::optional<u32>(Common::swap32(data)) : std::nullopt;
}
void RecordPPCEvent(Core::System* system, u32 pc)
{
auto& trace = s_ppc_event_trace;
if (!trace.output.IsOpen())
return;
const auto now = std::chrono::steady_clock::now();
const auto& ppc = system->GetPowerPC().GetPPCState();
std::string row =
fmt::format("{},{},{},{:08x},{:08x},{:08x}", ++trace.rows,
std::chrono::duration_cast<std::chrono::microseconds>(now - trace.start).count(),
system->GetCoreTiming().GetTicks(), pc, LR(ppc), CTR(ppc));
for (u32 reg = 3; reg <= 8; ++reg)
row += fmt::format(",{:08x}", ppc.gpr[reg]);
for (u32 word = 0; word < 7; ++word)
{
const auto value = PeekEventWord(*system, ppc.gpr[4] + word * 4);
row += value ? fmt::format(",{:08x}", *value) : ",NA";
}
row += '\n';
if (!trace.output.WriteString(row) || trace.rows >= 300000)
trace.output.Close();
else if (now - trace.last_flush >= std::chrono::milliseconds(250))
{
trace.output.Flush();
trace.last_flush = now;
}
}
} // namespace
// Dolphin's PowerPC->x86_64 JIT dynamic recompiler
// Written mostly by ector (hrydgard)
// Features:
@@ -255,6 +365,7 @@ bool Jit64::BackPatch(SContext* ctx)
void Jit64::Init()
{
s_ppc_event_trace.Init();
InitFastmemArena();
RefreshConfig();
@@ -338,6 +449,9 @@ void Jit64::ResetFreeMemoryRanges()
void Jit64::Shutdown()
{
if (s_ppc_event_trace.output.IsOpen())
s_ppc_event_trace.output.Close();
s_ppc_event_trace.points.clear();
FreeCodeSpace();
auto& memory = m_system.GetMemory();
@@ -824,6 +938,19 @@ void Jit64::Jit(u32 em_address, bool clear_cache_and_retry_on_failure)
}
}
if (!s_ppc_event_trace.points.empty())
{
// Force tracepoints to be block entries, where all guest registers are materialized. Avoid
// following branches past these boundaries. Ordinary runs retain their normal optimizations.
analyzer.ClearOption(PPCAnalyst::PPCAnalyzer::OPTION_BRANCH_FOLLOW);
const auto next = s_ppc_event_trace.points.lower_bound(em_address);
if (next != s_ppc_event_trace.points.end())
{
const size_t distance = (next->first - em_address) / 4;
block_size = std::min(block_size, distance == 0 ? size_t{1} : distance);
}
}
// Analyze the block, collect all instructions it is made of (including inlining,
// if that is enabled), reorder instructions for optimal performance, and join joinable
// instructions.
@@ -926,6 +1053,15 @@ bool Jit64::DoJit(u32 em_address, JitBlock* b, u32 nextPC)
// TODO: Test if this or AlignCode16 make a difference from GetCodePtr
b->normalEntry = AlignCode4();
const auto tracepoint = s_ppc_event_trace.points.find(em_address);
if (tracepoint != s_ppc_event_trace.points.end() &&
m_code_buffer[0].inst.hex == tracepoint->second)
{
ABI_PushRegistersAndAdjustStack({}, 0);
ABI_CallFunctionPC(RecordPPCEvent, &m_system, em_address);
ABI_PopRegistersAndAdjustStack({}, 0);
}
// Used to get a trace of the last few blocks before a crash, sometimes VERY useful
if (m_im_here_debug)
{
@@ -262,6 +262,10 @@ void Jit64AsmRoutineManager::GenerateCommon()
GenFres();
mfcr = AlignCode4();
GenMfcr();
dcache32_read_hit_dbat = AlignCode4();
GenDCache32Hit(false);
dcache32_write_hit_dbat = AlignCode4();
GenDCache32Hit(true);
cdts = AlignCode4();
GenConvertDoubleToSingle();
fmadds_eft = AlignCode4();
@@ -376,9 +376,53 @@ void EmuCodeBlock::SafeLoadToReg(X64Reg reg_value, const Gen::OpArg& opAddress,
LEA(32, RSCRATCH, MDisp(opAddress.GetSimpleReg(), offset));
}
FixupBranch exit;
const bool dr_set =
(flags & SAFE_LOADSTORE_DR_ON) || (m_jit.m_ppc_state.feature_flags & FEATURE_FLAG_MSR_DR);
FixupBranch accurate_dcache_done;
const bool accurate_dcache_hit_path =
!force_slow_access && dr_set && accessSize == 32 && !m_jit.jo.memcheck &&
m_jit.m_ppc_state.m_enable_dcache &&
(reg_addr != ABI_RETURN || (offset && opAddress.GetSimpleReg() != ABI_RETURN));
if (accurate_dcache_hit_path)
{
BitSet32 fast_registers_in_use = registersInUse;
if (reg_addr != ABI_RETURN)
fast_registers_in_use[reg_addr] = true;
else
fast_registers_in_use[opAddress.GetSimpleReg()] = true;
// Preserve scratch registers that remain live, including the original effective address
// needed by a cache-miss fallback. RDX can hold an address or paired-load intermediate even
// though the GPR allocator does not use it.
const bool preserve_rdx = fast_registers_in_use[RSCRATCH2];
const bool preserve_rcx = fast_registers_in_use[RSCRATCH_EXTRA];
if (preserve_rdx)
PUSH(RSCRATCH2);
if (preserve_rcx)
PUSH(RSCRATCH_EXTRA);
if (reg_addr != RSCRATCH)
MOV(32, R(RSCRATCH), R(reg_addr));
CALL(CommonAsmRoutines::dcache32_read_hit_dbat);
if (preserve_rcx)
POP(RSCRATCH_EXTRA);
if (preserve_rdx)
POP(RSCRATCH2);
TEST(64, R(ABI_RETURN), R(ABI_RETURN));
const FixupBranch slow = J_CC(CC_Z, Jump::Near);
SUB(64, R(ABI_RETURN), Imm8(1));
if (reg_value != ABI_RETURN)
MOV(32, R(reg_value), R(ABI_RETURN));
accurate_dcache_done = J(Jump::Near);
SetJumpTarget(slow);
// An offset address lives in RAX, which is also the ABI return register. Recreate it only on
// the miss path before calling the full helper.
if (reg_addr == ABI_RETURN && offset)
LEA(32, RSCRATCH, MDisp(opAddress.GetSimpleReg(), offset));
}
FixupBranch exit;
const bool fast_check_address =
!force_slow_access && dr_set && m_jit.jo.fastmem_arena && !m_jit.m_ppc_state.m_enable_dcache;
if (fast_check_address)
@@ -438,6 +482,9 @@ void EmuCodeBlock::SafeLoadToReg(X64Reg reg_value, const Gen::OpArg& opAddress,
}
SetJumpTarget(exit);
}
if (accurate_dcache_hit_path)
SetJumpTarget(accurate_dcache_done);
}
void EmuCodeBlock::SafeLoadToRegImmediate(X64Reg reg_value, u32 address, int accessSize,
@@ -533,12 +580,15 @@ void EmuCodeBlock::SafeWriteRegToReg(OpArg reg_value, X64Reg reg_addr, int acces
return;
}
const X64Reg original_reg_addr = reg_addr;
bool address_in_return_register = false;
if (offset)
{
if (flags & SAFE_LOADSTORE_CLOBBER_RSCRATCH_INSTEAD_OF_ADDR)
{
LEA(32, RSCRATCH, MDisp(reg_addr, (u32)offset));
reg_addr = RSCRATCH;
address_in_return_register = true;
}
else
{
@@ -546,9 +596,53 @@ void EmuCodeBlock::SafeWriteRegToReg(OpArg reg_value, X64Reg reg_addr, int acces
}
}
FixupBranch exit;
const bool dr_set =
(flags & SAFE_LOADSTORE_DR_ON) || (m_jit.m_ppc_state.feature_flags & FEATURE_FLAG_MSR_DR);
FixupBranch accurate_dcache_done;
const bool accurate_dcache_hit_path =
!force_slow_access && dr_set && accessSize == 32 && swap && !m_jit.jo.memcheck &&
m_jit.m_ppc_state.m_enable_dcache &&
(reg_addr != ABI_RETURN || (address_in_return_register && original_reg_addr != ABI_RETURN)) &&
(!reg_value.IsSimpleReg() || reg_value.GetSimpleReg() != ABI_RETURN);
if (accurate_dcache_hit_path)
{
BitSet32 fast_registers_in_use = registersInUse;
if (reg_addr != ABI_RETURN)
fast_registers_in_use[reg_addr] = true;
else
fast_registers_in_use[original_reg_addr] = true;
if (reg_value.IsSimpleReg())
fast_registers_in_use[reg_value.GetSimpleReg()] = true;
// The value setup itself overwrites RDX, so save it before loading the call arguments.
// On a miss, the full MMU helper must see the original address and value registers.
const bool preserve_rdx = fast_registers_in_use[RSCRATCH2];
const bool preserve_rcx = fast_registers_in_use[RSCRATCH_EXTRA];
if (preserve_rdx)
PUSH(RSCRATCH2);
if (preserve_rcx)
PUSH(RSCRATCH_EXTRA);
// Move the address first because it is allowed to arrive in RDX, which is also the native
// routine's value input.
if (reg_addr != RSCRATCH)
MOV(32, R(RSCRATCH), R(reg_addr));
if (reg_value.IsImm())
MOV(32, R(RSCRATCH2), reg_value);
else if (reg_value.GetSimpleReg() != RSCRATCH2)
MOV(32, R(RSCRATCH2), reg_value);
CALL(CommonAsmRoutines::dcache32_write_hit_dbat);
if (preserve_rcx)
POP(RSCRATCH_EXTRA);
if (preserve_rdx)
POP(RSCRATCH2);
TEST(8, R(ABI_RETURN), R(ABI_RETURN));
accurate_dcache_done = J_CC(CC_NZ, Jump::Near);
if (address_in_return_register)
LEA(32, RSCRATCH, MDisp(original_reg_addr, (u32)offset));
}
FixupBranch exit;
const bool fast_check_address =
!force_slow_access && dr_set && m_jit.jo.fastmem_arena && !m_jit.m_ppc_state.m_enable_dcache;
if (fast_check_address)
@@ -615,6 +709,9 @@ void EmuCodeBlock::SafeWriteRegToReg(OpArg reg_value, X64Reg reg_addr, int acces
}
SetJumpTarget(exit);
}
if (accurate_dcache_hit_path)
SetJumpTarget(accurate_dcache_done);
}
void EmuCodeBlock::SafeWriteRegToReg(Gen::X64Reg reg_value, Gen::X64Reg reg_addr, int accessSize,
@@ -14,8 +14,12 @@
#include "Common/x64ABI.h"
#include "Common/x64Emitter.h"
#include "Core/PowerPC/Gekko.h"
#include "Core/PowerPC/Jit64/Jit.h"
#include "Core/PowerPC/Jit64Common/Jit64Constants.h"
#include "Core/PowerPC/Jit64Common/Jit64PowerPCState.h"
#include "Core/PowerPC/MMU.h"
#include "Core/PowerPC/PPCCache.h"
#include "Core/System.h"
#define QUANTIZED_REGS_TO_SAVE \
(ABI_ALL_CALLER_SAVED & ~BitSet32{RSCRATCH, RSCRATCH2, RSCRATCH_EXTRA, XMM0 + 16, XMM1 + 16})
@@ -24,11 +28,166 @@
using namespace Gen;
const u8* CommonAsmRoutines::dcache32_read_hit_dbat = nullptr;
const u8* CommonAsmRoutines::dcache32_write_hit_dbat = nullptr;
alignas(16) static const __m128i double_fraction = _mm_set_epi64x(0, 0x000fffffffffffff);
alignas(16) static const __m128i double_explicit_top_bit = _mm_set_epi64x(0, 0x0010000000000000);
alignas(16) static const __m128i double_top_two_bits = _mm_set_epi64x(0, 0xc000000000000000);
alignas(16) static const __m128i double_bottom_bits = _mm_set_epi64x(0, 0x07ffffffe0000000);
constexpr std::array<u8, 128 * PowerPC::CACHE_WAYS> s_dcache_plru_update = [] {
std::array<u8, 128 * PowerPC::CACHE_WAYS> result{};
for (u32 old_plru = 0; old_plru < 128; ++old_plru)
{
for (u32 way = 0; way < PowerPC::CACHE_WAYS; ++way)
{
result[old_plru * PowerPC::CACHE_WAYS + way] =
(old_plru & ~PowerPC::Cache::PLRU_MASK[way]) | PowerPC::Cache::PLRU_VALUE[way];
}
}
return result;
}();
void CommonAsmRoutines::GenDCache32Hit(const bool write)
{
// This is the native equivalent of MMU::TryRead/WriteDCache32ForJit for the BAT-mapped
// MEM1/MEM2 hit case. Keeping it as a shared leaf routine avoids both the large C++ ABI
// register save and duplicating this sequence at every guest load/store.
//
// read: EAX = effective address; RAX = byte-swapped value + 1, or zero on miss
// write: EAX = effective address, EDX = guest value; EAX = one on hit, zero on miss
// Only the three Jit64 scratch registers are clobbered.
const void* const start = GetCodePtr();
auto& cache = m_jit.m_ppc_state.dCache;
auto& memory = m_jit.m_system.GetMemory();
ASSERT(!cache.lookup_table.empty());
ASSERT(!memory.GetEXRAM() || !cache.lookup_table_ex.empty());
// A write needs RDX for address translation, so keep its input value below the return address.
if (write)
PUSH(RSCRATCH2);
MOV(32, R(RSCRATCH2), R(RSCRATCH));
SHR(32, R(RSCRATCH2), Imm8(PowerPC::BAT_INDEX_SHIFT));
MOV(64, R(RSCRATCH_EXTRA), ImmPtr(m_jit.m_mmu.GetDBATTable().data()));
MOV(32, R(RSCRATCH2), MComplex(RSCRATCH_EXTRA, RSCRATCH2, SCALE_4, 0));
MOV(32, R(RSCRATCH_EXTRA), R(RSCRATCH2));
AND(32, R(RSCRATCH_EXTRA), Imm32(PowerPC::BAT_MAPPED_BIT | PowerPC::BAT_WI_BIT));
CMP(32, R(RSCRATCH_EXTRA), Imm32(PowerPC::BAT_MAPPED_BIT));
const FixupBranch invalid_bat = J_CC(CC_NE, Jump::Near);
AND(32, R(RSCRATCH2), Imm32(PowerPC::BAT_RESULT_MASK));
AND(32, R(RSCRATCH), Imm32(PowerPC::BAT_PAGE_SIZE - 1));
OR(32, R(RSCRATCH2), R(RSCRATCH));
// Select the lookup table while normalizing the physical address. Cache set and byte offset
// are identical for MEM1 and MEM2 after this normalization.
TEST(32, R(RSCRATCH2), Imm32(0xF8000000));
const FixupBranch mem1 = J_CC(CC_Z, Jump::Near);
MOV(32, R(RSCRATCH), R(RSCRATCH2));
AND(32, R(RSCRATCH), Imm32(0xF0000000));
CMP(32, R(RSCRATCH), Imm32(0x10000000));
const FixupBranch not_exram = J_CC(CC_NE, Jump::Near);
AND(32, R(RSCRATCH2), Imm32(0x0FFFFFFF));
CMP(32, R(RSCRATCH2), Imm32(memory.GetEXRAM() ? memory.GetExRamSizeReal() : 0));
const FixupBranch exram_out_of_range = J_CC(CC_AE, Jump::Near);
MOV(64, R(RSCRATCH), ImmPtr(cache.lookup_table_ex.data()));
const FixupBranch lookup_ready = J(Jump::Near);
SetJumpTarget(mem1);
AND(32, R(RSCRATCH2), Imm32(memory.GetRamMask()));
MOV(64, R(RSCRATCH), ImmPtr(cache.lookup_table.data()));
SetJumpTarget(lookup_ready);
MOV(32, R(RSCRATCH_EXTRA), R(RSCRATCH2));
SHR(32, R(RSCRATCH_EXTRA), Imm8(5));
MOVZX(32, 8, RSCRATCH_EXTRA, MComplex(RSCRATCH, RSCRATCH_EXTRA, SCALE_1, 0));
CMP(32, R(RSCRATCH_EXTRA), Imm32(0xff));
const FixupBranch cache_miss = J_CC(CC_E, Jump::Near);
// The generic helper preserves split cache-line transactions. Ordinary aligned 32-bit PPC
// traffic never takes this branch.
MOV(32, R(RSCRATCH), R(RSCRATCH2));
AND(32, R(RSCRATCH), Imm32(31));
CMP(32, R(RSCRATCH), Imm32(28));
const FixupBranch split_access = J_CC(CC_A, Jump::Near);
// RDX becomes a compact byte index into Cache::data:
// set * (8 ways * 32 bytes) + way * 32 + byte offset.
MOV(32, R(RSCRATCH), R(RSCRATCH2));
AND(32, R(RSCRATCH), Imm32(0xFE0));
SHL(32, R(RSCRATCH), Imm8(3));
AND(32, R(RSCRATCH2), Imm32(31));
OR(32, R(RSCRATCH2), R(RSCRATCH));
SHL(32, R(RSCRATCH_EXTRA), Imm8(5));
OR(32, R(RSCRATCH2), R(RSCRATCH_EXTRA));
// Update the exact 8-way pseudo-LRU state. Preserve the compact data index across the lookup.
PUSH(RSCRATCH2);
MOV(32, R(RSCRATCH), R(RSCRATCH2));
SHR(32, R(RSCRATCH), Imm8(8));
AND(32, R(RSCRATCH), Imm32(PowerPC::CACHE_SETS - 1));
MOV(32, R(RSCRATCH_EXTRA), R(RSCRATCH2));
SHR(32, R(RSCRATCH_EXTRA), Imm8(5));
AND(32, R(RSCRATCH_EXTRA), Imm32(PowerPC::CACHE_WAYS - 1));
MOV(64, R(RSCRATCH2), ImmPtr(cache.plru.data()));
MOVZX(32, 8, RSCRATCH2, MComplex(RSCRATCH2, RSCRATCH, SCALE_1, 0));
SHL(32, R(RSCRATCH2), Imm8(3));
OR(32, R(RSCRATCH2), R(RSCRATCH_EXTRA));
MOV(64, R(RSCRATCH_EXTRA), ImmPtr(s_dcache_plru_update.data()));
MOVZX(32, 8, RSCRATCH2, MComplex(RSCRATCH_EXTRA, RSCRATCH2, SCALE_1, 0));
MOV(64, R(RSCRATCH_EXTRA), ImmPtr(cache.plru.data()));
MOV(8, MComplex(RSCRATCH_EXTRA, RSCRATCH, SCALE_1, 0), R(RSCRATCH2));
POP(RSCRATCH2);
if (write)
{
// Restore, byte-swap and store the guest value, then mark the line dirty.
POP(RSCRATCH);
BSWAP(32, RSCRATCH);
MOV(64, R(RSCRATCH_EXTRA), ImmPtr(cache.data.data()));
MOV(32, MComplex(RSCRATCH_EXTRA, RSCRATCH2, SCALE_1, 0), R(RSCRATCH));
MOV(32, R(RSCRATCH), R(RSCRATCH2));
SHR(32, R(RSCRATCH), Imm8(8));
AND(32, R(RSCRATCH), Imm32(PowerPC::CACHE_SETS - 1));
MOV(32, R(RSCRATCH_EXTRA), R(RSCRATCH2));
SHR(32, R(RSCRATCH_EXTRA), Imm8(5));
AND(32, R(RSCRATCH_EXTRA), Imm32(PowerPC::CACHE_WAYS - 1));
MOV(32, R(RSCRATCH2), Imm32(1));
SHL(32, R(RSCRATCH2), R(RSCRATCH_EXTRA));
MOV(64, R(RSCRATCH_EXTRA), ImmPtr(cache.modified.data()));
OR(8, MComplex(RSCRATCH_EXTRA, RSCRATCH, SCALE_1, 0), R(RSCRATCH2));
MOV(32, R(RSCRATCH), Imm32(1));
RET();
}
else
{
MOV(64, R(RSCRATCH_EXTRA), ImmPtr(cache.data.data()));
MOV(32, R(RSCRATCH), MComplex(RSCRATCH_EXTRA, RSCRATCH2, SCALE_1, 0));
BSWAP(32, RSCRATCH);
ADD(64, R(RSCRATCH), Imm8(1));
RET();
}
SetJumpTarget(invalid_bat);
SetJumpTarget(not_exram);
SetJumpTarget(exram_out_of_range);
SetJumpTarget(cache_miss);
SetJumpTarget(split_access);
if (write)
POP(RSCRATCH2);
XOR(32, R(RSCRATCH), R(RSCRATCH));
RET();
if (write)
Common::JitRegister::Register(start, GetCodePtr(), "JIT_dcache32_write_hit");
else
Common::JitRegister::Register(start, GetCodePtr(), "JIT_dcache32_read_hit");
}
// Since the following float conversion functions are used in non-arithmetic PPC float
// instructions, they must convert floats bitexact and never flush denormals to zero or turn SNaNs
// into QNaNs. This means we can't use CVTSS2SD/CVTSD2SS.
@@ -27,9 +27,16 @@ class CommonAsmRoutines : public CommonAsmRoutinesBase, public QuantizedMemoryRo
{
public:
explicit CommonAsmRoutines(Jit64& jit) : QuantizedMemoryRoutines(jit) {}
// Accurate Broadway D-cache leaf routines. Static storage intentionally keeps the layout of
// Jit64AsmRoutineManager (and therefore Jit64) unchanged.
static const u8* dcache32_read_hit_dbat;
static const u8* dcache32_write_hit_dbat;
void GenFrsqrte();
void GenFres();
void GenMfcr();
void GenDCache32Hit(bool write);
protected:
void GenConvertDoubleToSingle();
+229 -2
View File
@@ -50,6 +50,7 @@
#include "Core/HW/MMIO.h"
#include "Core/HW/Memmap.h"
#include "Core/HW/ProcessorInterface.h"
#include "Core/HW/VideoInterface.h"
#include "Core/HW/WII_IPC.h"
#include "Core/IOS/Starlet/Starlet.h"
#include "Core/IOS/Starlet/StarletMemory.h"
@@ -250,6 +251,17 @@ T MMU::ReadFromHardware(u32 em_address)
}
}
// The Wii Menu polls VI_VERTICAL_BEAM_POSITION in a very tight loop while synchronizing its
// startup screens. Keep the exact live VI value, but avoid routing every 16-bit read through the
// generic MMIO mapping, type-erased handler and std::function layers. Address translation and
// all timing updates still happen normally before this point.
if constexpr (flag == XCheckTLBFlag::Read && sizeof(T) == sizeof(u16))
{
constexpr u32 vi_vertical_beam_position = 0x0c002000 | VideoInterface::VI_VERTICAL_BEAM_POSITION;
if (em_address == vi_vertical_beam_position)
return static_cast<T>(m_system.GetVideoInterface().GetVerticalBeamPosition());
}
if (flag == XCheckTLBFlag::Read && (em_address & 0xF8000000) == 0x08000000)
{
if (em_address < 0x0c000000)
@@ -821,6 +833,213 @@ void MMU::Write<u64>(const u64 var, const u32 address)
WriteToHardware<XCheckTLBFlag::Write>(address + sizeof(u32), static_cast<u32>(var), 4);
}
template <std::unsigned_integral T>
T MMU::ReadForJit(const u32 address)
{
if (!m_ppc_state.m_enable_dcache || m_power_pc.GetMemChecks().HasAny() ||
(address & HW_PAGE_MASK) > HW_PAGE_SIZE - sizeof(T))
{
return Read<T>(address);
}
u32 physical_address = address;
if (m_ppc_state.msr.DR)
{
const u32 bat_result = m_dbat_table[address >> BAT_INDEX_SHIFT];
if ((bat_result & BAT_MAPPED_BIT) == 0 || (bat_result & BAT_WI_BIT) != 0)
return Read<T>(address);
physical_address =
(bat_result & BAT_RESULT_MASK) | (address & (BAT_PAGE_SIZE - 1));
}
if (m_memory.GetRAM() && (physical_address & 0xF8000000) == 0)
{
physical_address &= m_memory.GetRamMask();
const T value = m_ppc_state.dCache.ReadMainMemoryValue<T, false>(
m_memory, physical_address, HID0(m_ppc_state).DLOCK);
return Common::FromBigEndian(value);
}
if (m_memory.GetEXRAM() && (physical_address >> 28) == 1 &&
(physical_address & 0x0FFFFFFF) < m_memory.GetExRamSizeReal())
{
physical_address = (physical_address & 0x0FFFFFFF) + CACHE_EXRAM_BIT;
const T value = m_ppc_state.dCache.ReadMainMemoryValue<T, true>(
m_memory, physical_address, HID0(m_ppc_state).DLOCK);
return Common::FromBigEndian(value);
}
return Read<T>(address);
}
template u8 MMU::ReadForJit<u8>(u32 address);
template u16 MMU::ReadForJit<u16>(u32 address);
template u32 MMU::ReadForJit<u32>(u32 address);
template u64 MMU::ReadForJit<u64>(u32 address);
template <std::unsigned_integral T>
void MMU::WriteForJit(const Common::MakeAtLeastU32<T> var, const u32 address)
{
// A 64-bit Broadway store is represented by two 32-bit bus writes in the generic path. Keep
// that uncommon operation there; byte, halfword and word animation/state stores use this path.
if constexpr (sizeof(T) > sizeof(u32))
{
Write<T>(var, address);
return;
}
if (!m_ppc_state.m_enable_dcache || m_power_pc.GetMemChecks().HasAny() ||
(address & HW_PAGE_MASK) > HW_PAGE_SIZE - sizeof(T))
{
Write<T>(var, address);
return;
}
u32 physical_address = address;
if (m_ppc_state.msr.DR)
{
const u32 bat_result = m_dbat_table[address >> BAT_INDEX_SHIFT];
if ((bat_result & BAT_MAPPED_BIT) == 0 || (bat_result & BAT_WI_BIT) != 0)
{
Write<T>(var, address);
return;
}
physical_address =
(bat_result & BAT_RESULT_MASK) | (address & (BAT_PAGE_SIZE - 1));
}
const T swapped_value = Common::FromBigEndian(static_cast<T>(var));
if (m_memory.GetRAM() && (physical_address & 0xF8000000) == 0)
{
physical_address &= m_memory.GetRamMask();
m_ppc_state.dCache.WriteMainMemoryValue<T, false>(
m_memory, physical_address, swapped_value, HID0(m_ppc_state).DLOCK);
return;
}
if (m_memory.GetEXRAM() && (physical_address >> 28) == 1 &&
(physical_address & 0x0FFFFFFF) < m_memory.GetExRamSizeReal())
{
physical_address = (physical_address & 0x0FFFFFFF) + CACHE_EXRAM_BIT;
m_ppc_state.dCache.WriteMainMemoryValue<T, true>(
m_memory, physical_address, swapped_value, HID0(m_ppc_state).DLOCK);
return;
}
Write<T>(var, address);
}
template void MMU::WriteForJit<u8>(u32 var, u32 address);
template void MMU::WriteForJit<u16>(u32 var, u32 address);
template void MMU::WriteForJit<u32>(u32 var, u32 address);
template void MMU::WriteForJit<u64>(u64 var, u32 address);
u64 MMU::TryReadDCache32ForJit(const u32 address)
{
// The generic helper handles the rare split transaction precisely.
if ((address & HW_PAGE_MASK) > HW_PAGE_SIZE - sizeof(u32))
return 0;
u32 physical_address = address;
if (m_ppc_state.msr.DR)
{
const u32 bat_result = m_dbat_table[address >> BAT_INDEX_SHIFT];
if ((bat_result & (BAT_MAPPED_BIT | BAT_WI_BIT)) != BAT_MAPPED_BIT)
return 0;
physical_address =
(bat_result & BAT_RESULT_MASK) | (address & (BAT_PAGE_SIZE - 1));
}
bool exram = false;
u32 lookup_index;
if (m_memory.GetRAM() && (physical_address & 0xF8000000) == 0)
{
physical_address &= m_memory.GetRamMask();
lookup_index = physical_address >> 5;
}
else if (m_memory.GetEXRAM() && (physical_address >> 28) == 1 &&
(physical_address & 0x0FFFFFFF) < m_memory.GetExRamSizeReal())
{
exram = true;
physical_address = (physical_address & 0x0FFFFFFF) + CACHE_EXRAM_BIT;
lookup_index = (physical_address & 0x0FFFFFFF) >> 5;
}
else
{
return 0;
}
if ((physical_address & 31) > 32 - sizeof(u32))
return 0;
Cache& cache = m_ppc_state.dCache;
const u32 line_address = physical_address & ~31U;
const u32 set = (line_address >> 5) & 0x7f;
const u32 way = exram ? cache.lookup_table_ex[lookup_index] : cache.lookup_table[lookup_index];
if (way == 0xff)
return 0;
cache.plru[set] = (cache.plru[set] & ~Cache::PLRU_MASK[way]) | Cache::PLRU_VALUE[way];
u32 value;
std::memcpy(&value,
reinterpret_cast<const u8*>(cache.data[set][way].data()) +
(physical_address & 31),
sizeof(value));
return static_cast<u64>(Common::swap32(value)) + 1;
}
bool MMU::TryWriteDCache32ForJit(const u32 var, const u32 address)
{
if ((address & HW_PAGE_MASK) > HW_PAGE_SIZE - sizeof(u32))
return false;
u32 physical_address = address;
if (m_ppc_state.msr.DR)
{
const u32 bat_result = m_dbat_table[address >> BAT_INDEX_SHIFT];
if ((bat_result & (BAT_MAPPED_BIT | BAT_WI_BIT)) != BAT_MAPPED_BIT)
return false;
physical_address =
(bat_result & BAT_RESULT_MASK) | (address & (BAT_PAGE_SIZE - 1));
}
bool exram = false;
u32 lookup_index;
if (m_memory.GetRAM() && (physical_address & 0xF8000000) == 0)
{
physical_address &= m_memory.GetRamMask();
lookup_index = physical_address >> 5;
}
else if (m_memory.GetEXRAM() && (physical_address >> 28) == 1 &&
(physical_address & 0x0FFFFFFF) < m_memory.GetExRamSizeReal())
{
exram = true;
physical_address = (physical_address & 0x0FFFFFFF) + CACHE_EXRAM_BIT;
lookup_index = (physical_address & 0x0FFFFFFF) >> 5;
}
else
{
return false;
}
if ((physical_address & 31) > 32 - sizeof(u32))
return false;
Cache& cache = m_ppc_state.dCache;
const u32 line_address = physical_address & ~31U;
const u32 set = (line_address >> 5) & 0x7f;
const u32 way = exram ? cache.lookup_table_ex[lookup_index] : cache.lookup_table[lookup_index];
if (way == 0xff)
return false;
cache.plru[set] = (cache.plru[set] & ~Cache::PLRU_MASK[way]) | Cache::PLRU_VALUE[way];
const u32 swapped_value = Common::swap32(var);
std::memcpy(reinterpret_cast<u8*>(cache.data[set][way].data()) + (physical_address & 31),
&swapped_value, sizeof(swapped_value));
cache.modified[set] |= 1U << way;
return true;
}
void MMU::Write_U16_Swap(const u32 var, const u32 address)
{
Write<u16>((var & 0xFFFF0000) | Common::swap16(static_cast<u16>(var)), address);
@@ -2183,10 +2402,18 @@ void ClearDCacheLineFromJit(MMU& mmu, u32 address)
{
mmu.ClearDCacheLine(address);
}
u64 TryReadDCache32FromJit(MMU& mmu, u32 address)
{
return mmu.TryReadDCache32ForJit(address);
}
bool TryWriteDCache32FromJit(MMU& mmu, u32 var, u32 address)
{
return mmu.TryWriteDCache32ForJit(var, address);
}
template <std::unsigned_integral T>
Common::MakeAtLeastU32<T> ReadFromJit(MMU& mmu, u32 address)
{
return mmu.Read<T>(address);
return mmu.ReadForJit<T>(address);
}
template u32 ReadFromJit<u8>(MMU& mmu, u32 address);
template u32 ReadFromJit<u16>(MMU& mmu, u32 address);
@@ -2196,7 +2423,7 @@ template u64 ReadFromJit<u64>(MMU& mmu, u32 address);
template <std::unsigned_integral T>
void WriteFromJit(MMU& mmu, Common::MakeAtLeastU32<T> var, u32 address)
{
mmu.Write<T>(var, address);
mmu.WriteForJit<T>(var, address);
}
template void WriteFromJit<u8>(MMU& mmu, u32 var, u32 address);
template void WriteFromJit<u16>(MMU& mmu, u32 var, u32 address);
+17
View File
@@ -233,6 +233,21 @@ public:
template <std::unsigned_integral T>
void Write(Common::MakeAtLeastU32<T> var, u32 address);
// JIT memory helpers. When accurate D-cache emulation is enabled, ordinary fastmem cannot be
// used because Broadway loads and stores must update the emulated cache rather than backing RAM.
// These helpers retain those exact semantics while bypassing the generic hardware dispatcher for
// the overwhelmingly common BAT-mapped MEM1/MEM2 case.
template <std::unsigned_integral T>
T ReadForJit(u32 address);
template <std::unsigned_integral T>
void WriteForJit(Common::MakeAtLeastU32<T> var, u32 address);
// Leaf fast paths used by Jit64 when accurate D-cache emulation is active. A zero/false result
// means that the caller must use ReadForJit/WriteForJit (cache miss, MMIO, TLB, WI, etc.). Reads
// encode a successful 32-bit value as value+1 in 64 bits so every guest value remains representable.
u64 TryReadDCache32ForJit(u32 address);
bool TryWriteDCache32ForJit(u32 var, u32 address);
void Write_U16_Swap(u32 var, u32 address);
void Write_U32_Swap(u32 var, u32 address);
void Write_U64_Swap(u64 var, u32 address);
@@ -374,6 +389,8 @@ private:
};
void ClearDCacheLineFromJit(MMU& mmu, u32 address);
u64 TryReadDCache32FromJit(MMU& mmu, u32 address);
bool TryWriteDCache32FromJit(MMU& mmu, u32 var, u32 address);
template <std::unsigned_integral T>
// Returns zero-extended value
Common::MakeAtLeastU32<T> ReadFromJit(MMU& mmu, u32 address);
+28 -6
View File
@@ -734,10 +734,11 @@ void PPCAnalyzer::SetInstructionStats(CodeBlock* block, CodeOp* code,
}
}
bool PPCAnalyzer::IsBusyWaitLoop(CodeBlock* block, CodeOp* code, size_t instructions) const
bool PPCAnalyzer::IsBusyWaitLoop(CodeOp* code, size_t loop_start, size_t branch_index) const
{
// Very basic algorithm to detect busy wait loops:
// * It loops to itself and does not contain any other branches.
// * It loops to the first instruction in the candidate range and does not contain any other
// branches.
// * It does not write to memory.
// * It only reads from registers it wrote to earlier in the loop, or it
// does not write to these registers.
@@ -748,14 +749,15 @@ bool PPCAnalyzer::IsBusyWaitLoop(CodeBlock* block, CodeOp* code, size_t instruct
// don't detect these at the moment.
std::bitset<32> write_disallowed_regs;
std::bitset<32> written_regs;
for (size_t i = 0; i <= instructions; ++i)
for (size_t i = loop_start; i <= branch_index; ++i)
{
if (code[i].opinfo->type == OpType::Branch)
{
if (code[i].branchUsesCtr)
return false;
if (code[i].branchTo == block->m_address && i == instructions)
if (code[i].branchTo == code[loop_start].address && i == branch_index)
return true;
return false;
}
// A `nop` is actually a `ori r0, r0, 0`, which would violate the rules (unless `r0` was written
// earlier).
@@ -763,6 +765,12 @@ bool PPCAnalyzer::IsBusyWaitLoop(CodeBlock* block, CodeOp* code, size_t instruct
{
continue;
}
// The JIT implements sync as a host no-op. It is commonly placed directly after an MMIO load
// in hardware polling loops, and does not make a load-only loop unsafe to idle.
else if (code[i].inst.OPCD == 31 && code[i].inst.SUBOP10 == 598)
{
continue;
}
else if (code[i].opinfo->type != OpType::Integer && code[i].opinfo->type != OpType::Load)
{
// In the future, some subsets of other instruction types might get
@@ -946,8 +954,22 @@ u32 PPCAnalyzer::Analyze(u32 address, CodeBlock* block, CodeBuffer* buffer,
}
}
code[i].branchIsIdleLoop =
code[i].branchTo == block->m_address && IsBusyWaitLoop(block, code, i);
code[i].branchIsIdleLoop = false;
if (code[i].opinfo->type == OpType::Branch)
{
// Branch following can inline a small polling function into its caller. Such a loop branches
// to an instruction in the middle of the analyzed block rather than block->m_address, so the
// old detector missed it. Locate the innermost matching instruction and validate only that
// loop range.
for (size_t candidate = i + 1; candidate-- > 0;)
{
if (code[candidate].address == code[i].branchTo)
{
code[i].branchIsIdleLoop = IsBusyWaitLoop(code, candidate, i);
break;
}
}
}
if (follow && numFollows < BRANCH_FOLLOWING_THRESHOLD)
{
+1 -1
View File
@@ -199,7 +199,7 @@ private:
ReorderType type) const;
void ReorderInstructions(u32 instructions, CodeOp* code) const;
void SetInstructionStats(CodeBlock* block, CodeOp* code, const GekkoOPInfo* opinfo) const;
bool IsBusyWaitLoop(CodeBlock* block, CodeOp* code, size_t instructions) const;
bool IsBusyWaitLoop(CodeOp* code, size_t loop_start, size_t branch_index) const;
// Options
u32 m_options = 0;
+5 -1
View File
@@ -275,10 +275,14 @@ void PowerPCManager::Init(CPUCore cpu_core)
Reset();
InitializeCPUCore(cpu_core);
auto& memory = m_system.GetMemory();
m_ppc_state.iCache.Init(memory);
m_ppc_state.dCache.Init(memory);
// JIT common routines may embed stable pointers into the cache lookup tables. Allocate those
// tables before generating a CPU core, rather than leaving the JIT with the empty vectors from
// PowerPCState construction. Reset and state loads only refill these allocations afterwards.
InitializeCPUCore(cpu_core);
}
void PowerPCManager::Reset()
+1 -1
View File
@@ -97,7 +97,7 @@ struct CompressAndDumpStateArgs
static Common::WorkQueueThreadSP<CompressAndDumpStateArgs> s_compress_and_dump_thread;
// Don't forget to increase this after doing changes on the savestate system
constexpr u32 STATE_VERSION = 192; // Last changed in PR 14646
constexpr u32 STATE_VERSION = 193; // Starlet HW_TIMER counter offset.
// Increase this if the StateExtendedHeader definition changes
constexpr u32 EXTENDED_HEADER_VERSION = 1; // Last changed in PR 12217
+1
View File
@@ -26,6 +26,7 @@ if(_M_X86_64)
PowerPC/DivUtilsTest.cpp
PowerPC/PageTableHostMappingTest.cpp
PowerPC/Jit64Common/ConvertDoubleToSingle.cpp
PowerPC/Jit64Common/DCache.cpp
PowerPC/Jit64Common/Fres.cpp
PowerPC/Jit64Common/Frsqrte.cpp
)
@@ -10,6 +10,7 @@
#include <gtest/gtest.h>
#include "Common/ChunkFile.h"
#include "Common/CommonTypes.h"
#include "Core/Core.h"
#include "Core/HW/WII_IPC.h"
@@ -171,6 +172,14 @@ public:
return m_sram_fastmem_enabled ? &m_sram_split_mode : nullptr;
}
const u8* GetDirectMemoryPointer(u32 address, u32 size) const override
{
const size_t offset = ToOffset(address);
if (size == 0 || offset > m_memory.size() || size > m_memory.size() - offset)
return nullptr;
return m_memory.data() + offset;
}
void SetIdlePollSafe(bool safe) { m_idle_poll_safe = safe; }
void SetSliceStablePollAddress(u32 address)
{
@@ -376,7 +385,7 @@ TEST(StarletTimer, ZeroDelayAlarmMatchesImmediatelyAndUsesIRQW1C)
memory.Write8(address + 3, static_cast<u8>(value));
};
memory.AdvanceCycles(405);
memory.AdvanceCycles(32 * 128);
ASSERT_EQ(read_word(timer), 32u);
write_word(alarm, read_word(timer));
EXPECT_EQ(system.GetWiiIPC().ReadStarletRegister(0x38) & INT_CAUSE_TIMER, INT_CAUSE_TIMER);
@@ -388,12 +397,155 @@ TEST(StarletTimer, ZeroDelayAlarmMatchesImmediatelyAndUsesIRQW1C)
write_word(arm_irq_flag, INT_CAUSE_TIMER);
EXPECT_EQ(system.GetWiiIPC().ReadStarletRegister(0x38) & INT_CAUSE_TIMER, 0u);
memory.AdvanceCycles(404);
memory.AdvanceCycles(32 * 128 - 1);
EXPECT_EQ(system.GetWiiIPC().ReadStarletRegister(0x38) & INT_CAUSE_TIMER, 0u);
memory.AdvanceCycles(1);
EXPECT_EQ(system.GetWiiIPC().ReadStarletRegister(0x38) & INT_CAUSE_TIMER, INT_CAUSE_TIMER);
}
TEST(StarletRegisters, WideTimerAndInterruptAccessesMatchHardwareSemantics)
{
constexpr u32 hardware_base = 0x0d800000;
constexpr u32 timer = hardware_base + 0x10;
constexpr u32 arm_irq_flag = hardware_base + 0x38;
constexpr u32 arm_irq_mask = hardware_base + 0x3c;
Core::DeclareAsCPUThread();
auto& system = Core::System::GetInstance();
system.GetWiiIPC().Reset();
StarletMemory memory(system);
memory.Reset();
memory.AdvanceCycles(32 * 128);
EXPECT_EQ(memory.Read32(timer), 32u);
memory.Write32(timer, 64);
EXPECT_EQ(memory.Read32(timer), 64u);
system.GetWiiIPC().SetStarletInterrupt(INT_CAUSE_TIMER, true);
EXPECT_EQ(memory.Read32(arm_irq_flag) & INT_CAUSE_TIMER, INT_CAUSE_TIMER);
memory.Write32(arm_irq_flag, INT_CAUSE_TIMER);
EXPECT_EQ(memory.Read32(arm_irq_flag) & INT_CAUSE_TIMER, 0u);
memory.Write32(arm_irq_mask, 0x800619ef);
EXPECT_EQ(memory.Read32(arm_irq_mask), 0x800619efu);
}
TEST(StarletTimer, RunsAtOneTickPer128ARMCycles)
{
constexpr u32 timer = 0x0d800010;
Core::DeclareAsCPUThread();
auto& system = Core::System::GetInstance();
system.GetWiiIPC().Reset();
StarletMemory memory(system);
memory.Reset();
memory.AdvanceCycles(127);
EXPECT_EQ(memory.Read32(timer), 0u);
memory.AdvanceCycles(1);
EXPECT_EQ(memory.Read32(timer), 1u);
memory.AdvanceCycles(243'000'000 - 128);
EXPECT_EQ(memory.Read32(timer), 1'898'437u);
memory.AdvanceCycles(243'000'000);
EXPECT_EQ(memory.Read32(timer), 3'796'875u);
}
TEST(StarletTimer, SchedulerSlicePartitionDoesNotChangeClock)
{
Core::DeclareAsCPUThread();
auto& system = Core::System::GetInstance();
StarletMemory memory(system);
memory.Reset();
constexpr u64 total_cycles = 243'000'000;
// Active, IPC, and idle scheduler slices must use the same clock, with no
// fractional timer ticks lost at the end of a slice.
for (const u64 slice : {256u, 4096u, 24300u})
{
memory.Reset();
for (u64 elapsed = 0; elapsed < total_cycles;)
{
const u64 step = std::min(slice, total_cycles - elapsed);
memory.AdvanceCycles(step);
elapsed += step;
}
EXPECT_EQ(memory.Read32(0x0d800010), 1'898'437u) << "slice=" << slice;
EXPECT_EQ(memory.GetCycles(), total_cycles);
}
}
TEST(StarletTimer, CounterWritesDoNotRewindPeripheralClock)
{
constexpr u32 timer = 0x0d800010;
Core::DeclareAsCPUThread();
auto& system = Core::System::GetInstance();
StarletMemory memory(system);
memory.Reset();
memory.AdvanceCycles(1025);
for (const u32 value : {1u, 0xffffffffu, 0u, 0x12345678u})
{
memory.Write32(timer, value);
EXPECT_EQ(memory.Read32(timer), value);
EXPECT_EQ(memory.GetCycles(), 1025u);
}
// Reprogramming HW_TIMER leaves the free-running /128 clock phase intact.
memory.AdvanceCycles(126);
EXPECT_EQ(memory.Read32(timer), 0x12345678u);
memory.AdvanceCycles(1);
EXPECT_EQ(memory.Read32(timer), 0x12345679u);
}
TEST(StarletTimer, ByteAssembledCounterWritesMatchWideWrites)
{
constexpr u32 timer = 0x0d800010;
Core::DeclareAsCPUThread();
auto& system = Core::System::GetInstance();
StarletMemory memory(system);
memory.Reset();
memory.AdvanceCycles(1280);
memory.Write8(timer, 0x12);
memory.Write8(timer + 1, 0x34);
memory.Write8(timer + 2, 0x56);
memory.Write8(timer + 3, 0x78);
EXPECT_EQ(memory.Read32(timer), 0x12345678u);
EXPECT_EQ(memory.GetCycles(), 1280u);
memory.AdvanceCycles(128);
EXPECT_EQ(memory.Read32(timer), 0x12345679u);
}
TEST(StarletTimer, AlarmFiresAcrossCounterWrap)
{
constexpr u32 timer = 0x0d800010;
constexpr u32 alarm = 0x0d800014;
Core::DeclareAsCPUThread();
auto& system = Core::System::GetInstance();
system.GetWiiIPC().Reset();
StarletMemory memory(system);
memory.Reset();
memory.Write32(timer, 0xfffffffe);
memory.Write32(alarm, 1);
memory.AdvanceCycles(3 * 128 - 1);
EXPECT_EQ(memory.Read32(timer), 0u);
EXPECT_EQ(system.GetWiiIPC().ReadStarletRegister(0x38) & INT_CAUSE_TIMER, 0u);
memory.AdvanceCycles(1);
EXPECT_EQ(memory.Read32(timer), 1u);
EXPECT_EQ(system.GetWiiIPC().ReadStarletRegister(0x38) & INT_CAUSE_TIMER, INT_CAUSE_TIMER);
}
TEST(StarletTimer, ResetClearsCounterOffset)
{
Core::DeclareAsCPUThread();
auto& system = Core::System::GetInstance();
StarletMemory memory(system);
memory.Reset();
memory.AdvanceCycles(512);
memory.Write32(0x0d800010, 0x12345678);
memory.Reset();
EXPECT_EQ(memory.GetCycles(), 0u);
EXPECT_EQ(memory.Read32(0x0d800010), 0u);
memory.AdvanceCycles(128);
EXPECT_EQ(memory.Read32(0x0d800010), 1u);
}
TEST(StarletNAND, HardwareResetPreservesProgrammedFlash)
{
Core::DeclareAsCPUThread();
@@ -426,6 +578,36 @@ TEST(StarletNAND, HardwareResetPreservesProgrammedFlash)
EXPECT_EQ(memory.Read32(sram), 0x12345678u);
}
TEST(StarletTimer, StateRoundTripPreservesOffsetAndDividerPhase)
{
constexpr u32 timer = 0x0d800010;
Core::DeclareAsCPUThread();
auto& system = Core::System::GetInstance();
StarletMemory memory(system);
memory.Reset();
memory.AdvanceCycles(1025);
memory.Write32(timer, 0x12345678);
std::vector<u8> state_buffer(1024 * 1024);
u8* state_pointer = state_buffer.data();
PointerWrap writer(&state_pointer, state_buffer.size(), PointerWrap::Mode::Write);
memory.DoState(writer);
ASSERT_TRUE(writer.IsWriteMode());
const size_t state_size = state_pointer - state_buffer.data();
memory.Reset();
state_pointer = state_buffer.data();
PointerWrap reader(&state_pointer, state_size, PointerWrap::Mode::Read);
memory.DoState(reader);
ASSERT_TRUE(reader.IsReadMode());
EXPECT_EQ(memory.GetCycles(), 1025u);
EXPECT_EQ(memory.Read32(timer), 0x12345678u);
memory.AdvanceCycles(126);
EXPECT_EQ(memory.Read32(timer), 0x12345678u);
memory.AdvanceCycles(1);
EXPECT_EQ(memory.Read32(timer), 0x12345679u);
}
TEST(StarletGPIO, InterruptFlagIsWriteOneToClear)
{
constexpr u32 hardware_base = 0x0d800000;
@@ -984,6 +1166,35 @@ TEST(StarletARMCore, JitCompilesDrainWriteBufferNatively)
EXPECT_EQ(core.GetJitFallbackInstructionCount(), 1u);
}
TEST(StarletARMCore, JitCompilesHotCP15MaintenanceNatively)
{
#if defined(_M_X86_64)
TestBus interpreter_bus;
TestBus jit_bus;
ARMCore interpreter(interpreter_bus);
ARMCore jit(jit_bus);
jit.SetJitEnabled(true);
const auto install_program = [](TestBus& bus) {
bus.WriteARM(0x00, 0xee033f10); // mcr p15, 0, r3, c3, c0, 0 (DACR)
bus.WriteARM(0x04, 0xee070f36); // mcr p15, 0, r0, c7, c6, 1
bus.WriteARM(0x08, 0xee070f3a); // mcr p15, 0, r0, c7, c10, 1
bus.WriteARM(0x0c, 0xeafffffe); // b .
};
install_program(interpreter_bus);
install_program(jit_bus);
interpreter.SetRegister(3, 0x55555555);
jit.SetRegister(3, 0x55555555);
ASSERT_EQ(interpreter.RunCycles(4), 4u);
ASSERT_EQ(jit.RunCycles(4), 4u);
EXPECT_EQ(jit.GetCP15State().domain_access_control,
interpreter.GetCP15State().domain_access_control);
EXPECT_EQ(jit.GetRegister(15), interpreter.GetRegister(15));
EXPECT_EQ(jit.GetJitFallbackInstructionCount(), 0u);
EXPECT_EQ(jit.GetJitNativeExecutedInstructions(), 4u);
#endif
}
TEST(StarletARMCore, JitDefersCP15CacheInvalidationUntilTheHostBlockReturns)
{
TestBus bus;
@@ -1041,18 +1252,324 @@ TEST(StarletARMCore, JitPreservedBlocksUseCurrentTLBGenerationForFastmem)
EXPECT_EQ(core.RunCycles(2), 2u);
ASSERT_EQ(core.GetRegister(1), 0x11223344u);
ASSERT_EQ(core.GetJitFallbackInstructionCount(), 1u);
ASSERT_EQ(core.GetJitFallbackInstructionCount(), 0u);
ASSERT_EQ(core.GetJitCompiledBlockCount(), 1u);
core.SetRegister(1, 0);
core.SetRegister(15, 0x80000000);
EXPECT_EQ(core.RunCycles(1), 1u);
EXPECT_EQ(core.GetRegister(1), 0x11223344u);
EXPECT_EQ(core.GetJitFallbackInstructionCount(), 1u);
EXPECT_EQ(core.GetJitFallbackInstructionCount(), 0u);
EXPECT_EQ(core.GetJitCompiledBlockCount(), 1u);
#endif
}
TEST(StarletARMCore, JitTLBRevalidationIsSharedByNativeBlocksOnTheSamePage)
{
#if defined(_M_X86_64)
TestBus bus(0x10000);
ARMCore core(bus);
bus.WriteARM(0x0000, 0xe3a01001); // mov r1, #1
bus.WriteARM(0x0020, 0xe3a02002); // mov r2, #2
bus.WriteARM(0x0040, 0xee080f17); // invalidate unified TLB
bus.WriteARM(0x6000, 0x00000c02); // VA 0x80000000 section -> PA 0
core.GetCP15State().translation_table_base = 0x4000;
core.GetCP15State().domain_access_control = 3;
core.GetCP15State().control |= 1;
core.SetJitEnabled(true);
core.SetRegister(15, 0x80000000);
EXPECT_EQ(core.RunCycles(1), 1u);
core.SetRegister(15, 0x80000020);
EXPECT_EQ(core.RunCycles(1), 1u);
core.SetRegister(15, 0x80000040);
EXPECT_EQ(core.RunCycles(1), 1u);
const u64 dispatches_before_revalidation = core.GetJitDispatchSlowCount();
core.SetRegister(15, 0x80000000);
EXPECT_EQ(core.RunCycles(1), 1u);
EXPECT_EQ(core.GetJitDispatchSlowCount(), dispatches_before_revalidation);
// The first block revalidated the unchanged page-table descriptor directly in generated code.
// A second native block on that physical page must likewise avoid a page-table walk and C++
// block-map lookup.
core.SetRegister(15, 0x80000020);
EXPECT_EQ(core.RunCycles(1), 1u);
EXPECT_EQ(core.GetJitDispatchSlowCount(), dispatches_before_revalidation);
#endif
}
TEST(StarletARMCore, JitSharedTLBRefillRejectsRemappedPhysicalCode)
{
#if defined(_M_X86_64)
TestBus bus(0x110000);
ARMCore core(bus);
bus.WriteARM(0x000000, 0xe3a01001); // old page: mov r1, #1
bus.WriteARM(0x000020, 0xe3a02002); // old page: mov r2, #2
bus.WriteARM(0x000040, 0xee080f17); // invalidate unified TLB
bus.WriteARM(0x100000, 0xe3a01003); // new page: mov r1, #3
bus.WriteARM(0x100020, 0xe3a02004); // new page: mov r2, #4
bus.WriteARM(0x006000, 0x00000c02); // VA 0x80000000 section -> PA 0
core.GetCP15State().translation_table_base = 0x4000;
core.GetCP15State().domain_access_control = 3;
core.GetCP15State().control |= 1;
core.SetJitEnabled(true);
core.SetRegister(15, 0x80000000);
EXPECT_EQ(core.RunCycles(1), 1u);
core.SetRegister(15, 0x80000020);
EXPECT_EQ(core.RunCycles(1), 1u);
const size_t old_block_count = core.GetJitCompiledBlockCount();
// Change the page table under the still-valid TLB, then execute the architectural invalidation
// through the old mapping. The following dispatch must discover the new physical page.
bus.WriteARM(0x006000, 0x00100c02);
core.SetRegister(15, 0x80000040);
EXPECT_EQ(core.RunCycles(1), 1u);
core.SetRegister(15, 0x80000000);
EXPECT_EQ(core.RunCycles(1), 1u);
EXPECT_EQ(core.GetRegister(1), 3u);
core.SetRegister(15, 0x80000020);
EXPECT_EQ(core.RunCycles(1), 1u);
EXPECT_EQ(core.GetRegister(2), 4u);
EXPECT_EQ(core.GetJitCompiledBlockCount(), old_block_count + 3);
#endif
}
TEST(StarletARMCore, JitRetainsPhysicalAliasesAcrossAddressSpaceSwitches)
{
#if defined(_M_X86_64)
TestBus bus(0x110000);
ARMCore core(bus);
bus.WriteARM(0x000000, 0xe3a01001); // physical mapping 0: mov r1, #1
bus.WriteARM(0x000020, 0xe3a02002); // physical mapping 0: mov r2, #2
bus.WriteARM(0x000040, 0xee080f17); // invalidate unified TLB
bus.WriteARM(0x100000, 0xe3a01003); // physical mapping 1: mov r1, #3
bus.WriteARM(0x100020, 0xe3a02004); // physical mapping 1: mov r2, #4
bus.WriteARM(0x100040, 0xee080f17); // invalidate unified TLB
bus.WriteARM(0x006000, 0x00000c02); // VA 0x80000000 section -> PA 0
core.GetCP15State().translation_table_base = 0x4000;
core.GetCP15State().domain_access_control = 3;
core.GetCP15State().control |= 1;
core.SetJitEnabled(true);
for (const u32 address : {0x80000000U, 0x80000020U})
{
core.SetRegister(15, address);
EXPECT_EQ(core.RunCycles(1), 1u);
}
bus.WriteARM(0x006000, 0x00100c02); // Same virtual section -> PA 1 MiB.
core.SetRegister(15, 0x80000040);
EXPECT_EQ(core.RunCycles(1), 1u);
for (const u32 address : {0x80000000U, 0x80000020U})
{
core.SetRegister(15, address);
EXPECT_EQ(core.RunCycles(1), 1u);
}
// Return to the first address space. The first block refills the shared page translation; the
// second must immediately find its retained (MVA, physical page) entry instead of overwriting a
// single virtual-key slot and falling back to C++ again.
bus.WriteARM(0x006000, 0x00000c02);
core.SetRegister(15, 0x80000040);
EXPECT_EQ(core.RunCycles(1), 1u);
const u64 dispatches_before_refill = core.GetJitDispatchSlowCount();
core.SetRegister(15, 0x80000000);
EXPECT_EQ(core.RunCycles(1), 1u);
EXPECT_EQ(core.GetRegister(1), 1u);
core.SetRegister(15, 0x80000020);
EXPECT_EQ(core.RunCycles(1), 1u);
EXPECT_EQ(core.GetRegister(2), 2u);
EXPECT_EQ(core.GetJitDispatchSlowCount(), dispatches_before_refill + 1);
#endif
}
TEST(StarletARMCore, JitFastBlockCacheRetainsFourCollidingHotBlocks)
{
#if defined(_M_X86_64)
TestBus bus(0xd0000);
ARMCore core(bus);
core.SetJitEnabled(true);
// These ARM addresses deliberately have the same upper 16 bits after multiplying by the JIT
// cache's 0x9e3779b1 hash constant. They therefore occupy the four ways of one cache set.
constexpr std::array<u32, 5> addresses = {0x000014, 0x04cb94, 0x07e168, 0x099714, 0x0cace8};
for (const u32 address : addresses)
bus.WriteARM(address, 0xe3a01001); // mov r1, #1
for (size_t i = 0; i < 4; ++i)
{
const u32 address = addresses[i];
core.SetRegister(15, address);
EXPECT_EQ(core.RunCycles(1), 1u);
}
const u64 dispatches_after_fill = core.GetJitDispatchSlowCount();
const u64 collisions_after_fill = core.GetJitDispatchCollisionCount();
for (size_t i = 0; i < 4; ++i)
{
const u32 address = addresses[i];
core.SetRegister(15, address);
EXPECT_EQ(core.RunCycles(1), 1u);
}
EXPECT_EQ(core.GetJitDispatchSlowCount(), dispatches_after_fill);
EXPECT_EQ(core.GetJitDispatchCollisionCount(), collisions_after_fill);
// A fifth distinct key proves that the set is actually full and exercises bounded replacement.
core.SetRegister(15, addresses.back());
EXPECT_EQ(core.RunCycles(1), 1u);
EXPECT_EQ(core.GetJitDispatchSlowCount(), dispatches_after_fill + 1);
EXPECT_EQ(core.GetJitDispatchCollisionCount(), collisions_after_fill + 1);
#endif
}
TEST(StarletARMCore, JitCachesFallbackOnlyBlocks)
{
#if defined(_M_X86_64)
TestBus bus(0x10000);
ARMCore core(bus);
// MUL uses the exact interpreter helper in this JIT. A block beginning with it therefore has
// zero directly emitted ARM instructions, but its generated fallback wrapper is still reusable.
bus.WriteARM(0x0000, 0xe0010190); // mul r1, r0, r1
core.SetJitEnabled(true);
core.SetRegister(0, 3);
core.SetRegister(1, 4);
core.SetRegister(15, 0);
EXPECT_EQ(core.RunCycles(1), 1u);
EXPECT_EQ(core.GetRegister(1), 12u);
const u64 dispatches_after_compile = core.GetJitDispatchSlowCount();
const u64 fallbacks_after_compile = core.GetJitFallbackInstructionCount();
core.SetRegister(1, 5);
core.SetRegister(15, 0);
EXPECT_EQ(core.RunCycles(1), 1u);
EXPECT_EQ(core.GetRegister(1), 15u);
EXPECT_EQ(core.GetJitFallbackInstructionCount(), fallbacks_after_compile + 1);
EXPECT_EQ(core.GetJitDispatchSlowCount(), dispatches_after_compile);
#endif
}
TEST(StarletARMCore, ARMJitCompilesLogicalImmediateAndShiftCarry)
{
#if defined(_M_X86_64)
TestBus interpreter_bus(0x1000);
TestBus jit_bus(0x1000);
ARMCore interpreter(interpreter_bus);
ARMCore jit(jit_bus);
jit.SetJitEnabled(true);
const auto install_program = [](TestBus& bus) {
bus.WriteARM(0x00, 0xe3180701); // tst r8, #0x40000; rotated immediate supplies C
bus.WriteARM(0x04, 0xeafffffe); // b .
bus.WriteARM(0x20, 0xe1b02820); // movs r2, r0, lsr #16; bit 15 supplies C
bus.WriteARM(0x24, 0xeafffffe); // b .
};
install_program(interpreter_bus);
install_program(jit_bus);
for (ARMCore* core : {&interpreter, &jit})
{
core->SetCPSR(static_cast<u32>(ARMCore::Mode::System) | ARMCore::CPSR_C | ARMCore::CPSR_V);
core->SetRegister(8, 0x40000);
core->SetRegister(15, 0);
}
ASSERT_EQ(interpreter.RunCycles(2), 2u);
ASSERT_EQ(jit.RunCycles(2), 2u);
EXPECT_EQ(jit.GetCPSR(), interpreter.GetCPSR());
for (ARMCore* core : {&interpreter, &jit})
{
core->SetCPSR(static_cast<u32>(ARMCore::Mode::System) | ARMCore::CPSR_V);
core->SetRegister(0, 0x80018000);
core->SetRegister(2, 0);
core->SetRegister(15, 0x20);
}
ASSERT_EQ(interpreter.RunCycles(2), 2u);
ASSERT_EQ(jit.RunCycles(2), 2u);
EXPECT_EQ(jit.GetRegister(2), interpreter.GetRegister(2));
EXPECT_EQ(jit.GetCPSR(), interpreter.GetCPSR());
EXPECT_EQ(jit.GetJitFallbackInstructionCount(), 0u);
EXPECT_EQ(jit.GetJitNativeExecutedInstructions(), 4u);
#endif
}
TEST(StarletARMCore, ARMJitCompilesIRQVectorLoadPCWithInterworking)
{
#if defined(_M_X86_64)
TestBus bus;
ARMCore core(bus);
bus.SetFastmemEnabled(true);
bus.WriteARM(0x00, 0xe59ff018); // ldr pc, [pc, #0x18] -> 0x20
bus.WriteARM(0x20, 0x00000101); // enter Thumb at 0x100
core.SetRegister(15, 0);
core.SetJitEnabled(true);
EXPECT_EQ(core.RunCycles(1), 1u);
EXPECT_EQ(core.GetRegister(15), 0x100u);
EXPECT_NE(core.GetCPSR() & ARMCore::CPSR_T, 0u);
EXPECT_EQ(core.GetJitFallbackInstructionCount(), 0u);
EXPECT_EQ(core.GetJitSlowReadCount(), 0u);
EXPECT_EQ(core.GetJitNativeExecutedInstructions(), 1u);
EXPECT_TRUE(bus.SRAMCanariesIntact());
#endif
}
TEST(StarletARMCore, ARMJitCompilesLongMultiplyFamily)
{
#if defined(_M_X86_64)
struct Case
{
u32 instruction;
u32 cpsr;
u32 rm;
u32 rs;
u32 rd_hi;
u32 rd_lo;
};
constexpr std::array cases = {
Case{0xe0834291, static_cast<u32>(ARMCore::Mode::System) | ARMCore::CPSR_C, 0x10000, 0x10001,
0, 0}, // UMULL
Case{0xe0c34291, static_cast<u32>(ARMCore::Mode::System) | ARMCore::CPSR_V, 0xfffffff0, 0x10,
0, 0}, // SMULL
Case{0xe0b34291, static_cast<u32>(ARMCore::Mode::System) | ARMCore::CPSR_C | ARMCore::CPSR_V,
0xffffffff, 1, 0, 1}, // UMLALS, result wraps to zero and preserves CV
Case{0x10834291, static_cast<u32>(ARMCore::Mode::System) | ARMCore::CPSR_Z, 7, 9, 0x11223344,
0x55667788}, // UMULLNE, predicate fails
};
for (const Case& test : cases)
{
TestBus interpreter_bus;
TestBus jit_bus;
ARMCore interpreter(interpreter_bus);
ARMCore jit(jit_bus);
interpreter_bus.WriteARM(0, test.instruction);
interpreter_bus.WriteARM(4, 0xeafffffe); // b .
jit_bus.WriteARM(0, test.instruction);
jit_bus.WriteARM(4, 0xeafffffe); // b .
for (ARMCore* core : {&interpreter, &jit})
{
core->SetCPSR(test.cpsr);
core->SetRegister(1, test.rm);
core->SetRegister(2, test.rs);
core->SetRegister(3, test.rd_hi);
core->SetRegister(4, test.rd_lo);
core->SetRegister(15, 0);
}
jit.SetJitEnabled(true);
ASSERT_EQ(interpreter.RunCycles(2), 2u);
ASSERT_EQ(jit.RunCycles(2), 2u);
EXPECT_EQ(jit.GetRegister(3), interpreter.GetRegister(3));
EXPECT_EQ(jit.GetRegister(4), interpreter.GetRegister(4));
EXPECT_EQ(jit.GetCPSR(), interpreter.GetCPSR());
EXPECT_EQ(jit.GetJitFallbackInstructionCount(), 0u);
EXPECT_EQ(jit.GetJitNativeExecutedInstructions(), 2u);
}
#endif
}
TEST(StarletARMCore, FCSESwitchPreservesTaggedTLBTranslations)
{
TestBus bus(0x10000);
@@ -0,0 +1,316 @@
// Copyright 2026 Dolphin Emulator Project
// SPDX-License-Identifier: GPL-2.0-or-later
#include <array>
#include <memory>
#include "Common/x64ABI.h"
#include "Core/ConfigManager.h"
#include "Core/Core.h"
#include "Core/HW/Memmap.h"
#include "Core/PowerPC/Jit64/Jit.h"
#include "Core/PowerPC/Jit64Common/Jit64AsmCommon.h"
#include "Core/PowerPC/Jit64Common/Jit64Constants.h"
#include "Core/PowerPC/MMU.h"
#include "Core/PowerPC/PowerPC.h"
#include "Core/System.h"
#include <gtest/gtest.h>
namespace
{
using namespace Gen;
// Execute the real SafeLoad/SafeWrite emitter, including its fallback and register contract.
// Testing only the C++ MMU helpers cannot detect corruption introduced at the native call site.
class DCacheCode : public CommonAsmRoutines
{
public:
explicit DCacheCode(Core::System& system) : CommonAsmRoutines(jit), jit(system)
{
jit.jo = {};
jit.js = {};
AllocCodeSpace(512 * 1024);
old_read = dcache32_read_hit_dbat;
old_write = dcache32_write_hit_dbat;
dcache32_read_hit_dbat = AlignCode4();
GenDCache32Hit(false);
dcache32_write_hit_dbat = AlignCode4();
GenDCache32Hit(true);
}
~DCacheCode() override
{
dcache32_read_hit_dbat = old_read;
dcache32_write_hit_dbat = old_write;
}
void Access(bool write, u32 address, u32 value, X64Reg address_reg, X64Reg value_reg,
s32 offset = 0, int flags = 0)
{
run = reinterpret_cast<void (*)()>(AlignCode4());
ABI_PushRegistersAndAdjustStack(ABI_ALL_CALLEE_SAVED, 8);
MOV(64, R(RPPCSTATE), ImmPtr(reinterpret_cast<u8*>(&jit.m_ppc_state) + 0x80));
MOV(32, R(RDX), Imm32(0x12345678));
MOV(32, R(RCX), Imm32(0x87654321));
MOV(32, R(address_reg), Imm32(address));
if (write)
MOV(32, R(value_reg), Imm32(value));
const BitSet32 live{RCX, RDX, R8, R9};
if (flags & SAFE_LOADSTORE_NO_PROLOG)
SUB(64, R(RSP), Imm8(8));
if (write)
SafeWriteRegToReg(value_reg, address_reg, 32, offset, live, flags);
else
SafeLoadToReg(value_reg, R(address_reg), 32, offset, live, false, flags);
if (flags & SAFE_LOADSTORE_NO_PROLOG)
ADD(64, R(RSP), Imm8(8));
MOV(64, R(R11), ImmPtr(result.data()));
for (const auto reg : {RAX, RCX, RDX, R8, R9})
MOV(64, MDisp(R11, static_cast<int>(reg) * sizeof(u64)), R(reg));
ABI_PopRegistersAndAdjustStack(ABI_ALL_CALLEE_SAVED, 8);
RET();
ASSERT_FALSE(HasWriteFailed());
run();
}
std::array<u64, 16> result{};
void (*run)() = nullptr;
Jit64 jit;
const u8* old_read;
const u8* old_write;
};
class Jit64DCache : public testing::Test
{
protected:
static void SetUpTestSuite()
{
SConfig::Init();
auto& system = Core::System::GetInstance();
system.SetIsWii(true);
system.GetMemory().Init();
Core::DeclareAsCPUThread();
system.GetPPCState().dCache.Init(system.GetMemory());
}
static void TearDownTestSuite()
{
auto& system = Core::System::GetInstance();
system.GetPPCState().m_enable_dcache = false;
system.GetMemory().Shutdown();
system.SetIsWii(false);
Core::UndeclareAsCPUThread();
SConfig::Shutdown();
}
void SetUp() override
{
state.dCache.Reset();
state.m_enable_dcache = true;
state.msr.DR = 1;
state.feature_flags = FEATURE_FLAG_MSR_DR;
state.Exceptions = 0;
state.spr[SPR_HID0] = 0;
bats.fill(0);
bats[0x80000000 >> PowerPC::BAT_INDEX_SHIFT] = PowerPC::BAT_MAPPED_BIT;
bats[0x90000000 >> PowerPC::BAT_INDEX_SHIFT] = 0x10000000 | PowerPC::BAT_MAPPED_BIT;
code = std::make_unique<DCacheCode>(system);
}
void TearDown() override { code.reset(); }
Core::System& system = Core::System::GetInstance();
PowerPC::PowerPCState& state = system.GetPPCState();
PowerPC::MMU& mmu = system.GetMMU();
Memory::MemoryManager& memory = system.GetMemory();
PowerPC::BatTable& bats = const_cast<PowerPC::BatTable&>(mmu.GetDBATTable());
std::unique_ptr<DCacheCode> code;
};
TEST_F(Jit64DCache, StoreMissRetainsAddressInRDX)
{
// The data is deliberately also a mapped address: a broken fallback writes there instead.
memory.Write_U32(0, 0x1000);
memory.Write_U32(0, 0x4000);
code->Access(true, 0x80001000, 0x80004000, RDX, RCX);
EXPECT_EQ(mmu.ReadForJit<u32>(0x80001000), 0x80004000);
EXPECT_EQ(mmu.ReadForJit<u32>(0x80004000), 0U);
EXPECT_EQ(code->result[RDX], 0x80001000);
EXPECT_EQ(code->result[RCX], 0x80004000);
}
TEST_F(Jit64DCache, LoadMissRetainsAddressInRDX)
{
memory.Write_U32(0x89abcdef, 0x1000);
code->Access(false, 0x80001000, 0, RDX, R8);
EXPECT_EQ(code->result[R8], 0x89abcdef);
EXPECT_EQ(code->result[RDX], 0x80001000);
}
TEST_F(Jit64DCache, HitPreservesLiveScratchRegisters)
{
mmu.WriteForJit<u32>(0xffffffff, 0x80001000);
code->Access(false, 0x80001000, 0, R9, R8);
EXPECT_EQ(code->result[R8], 0xffffffff);
EXPECT_EQ(code->result[RDX], 0x12345678U);
EXPECT_EQ(code->result[RCX], 0x87654321U);
code->Access(true, 0x80001000, 0x11223344, R9, R8);
EXPECT_EQ(mmu.ReadForJit<u32>(0x80001000), 0x11223344U);
EXPECT_EQ(code->result[RDX], 0x12345678U);
EXPECT_EQ(code->result[RCX], 0x87654321U);
}
TEST_F(Jit64DCache, PhysicalAccessIgnoresDBAT)
{
state.msr.DR = 0;
state.feature_flags = CPUEmuFeatureFlags{};
bats[0] = 0x20000 | PowerPC::BAT_MAPPED_BIT;
mmu.WriteForJit<u32>(0x11111111, 0x1000);
mmu.WriteForJit<u32>(0x22222222, 0x21000);
code->Access(false, 0x1000, 0, R9, R8);
EXPECT_EQ(code->result[R8], 0x11111111U);
code->Access(true, 0x1000, 0x33333333, R9, R8);
EXPECT_EQ(mmu.ReadForJit<u32>(0x1000), 0x33333333U);
EXPECT_EQ(mmu.ReadForJit<u32>(0x21000), 0x22222222U);
}
TEST_F(Jit64DCache, LoadRegisterAndOffsetCombinations)
{
for (const bool hit : {false, true})
{
for (const auto address_reg : {RAX, RCX, RDX, R8, R9})
{
for (const auto value_reg : {RAX, RCX, RDX, R8, R9})
{
for (const s32 offset : {0, 4, -4})
{
SCOPED_TRACE(testing::Message()
<< hit << ' ' << address_reg << ' ' << value_reg << ' ' << offset);
state.dCache.Reset();
memory.Write_U32(0xfedcba98, 0x1020);
if (hit)
ASSERT_EQ(mmu.ReadForJit<u32>(0x80001020), 0xfedcba98);
code->Access(false, 0x80001020 - offset, 0, address_reg, value_reg, offset);
EXPECT_EQ(code->result[value_reg], 0xfedcba98);
}
}
}
}
}
TEST_F(Jit64DCache, StoreRegisterAndOffsetCombinations)
{
for (const bool hit : {false, true})
{
for (const auto address_reg : {RAX, RCX, RDX, R8, R9})
{
for (const auto value_reg : {RCX, RDX, R8, R9})
{
if (address_reg == value_reg)
continue;
for (const s32 offset : {0, 4, -4})
{
SCOPED_TRACE(testing::Message()
<< hit << ' ' << address_reg << ' ' << value_reg << ' ' << offset);
state.dCache.Reset();
memory.Write_U32(0, 0x1020);
if (hit)
ASSERT_EQ(mmu.ReadForJit<u32>(0x80001020), 0U);
code->Access(true, 0x80001020 - offset, 0xfedcba98, address_reg, value_reg, offset,
EmuCodeBlock::SAFE_LOADSTORE_CLOBBER_RSCRATCH_INSTEAD_OF_ADDR);
EXPECT_EQ(mmu.ReadForJit<u32>(0x80001020), 0xfedcba98);
EXPECT_EQ(code->result[value_reg], 0xfedcba98);
}
}
}
}
}
TEST_F(Jit64DCache, SplitLineAndPageAccess)
{
for (const u32 offset : {29U, 30U, 31U, 4093U, 4094U, 4095U})
{
SCOPED_TRACE(offset);
const u32 address = 0x80001000 + offset;
mmu.WriteForJit<u32>(0x11223344, address);
code->Access(false, address, 0, RDX, R8);
EXPECT_EQ(code->result[R8], 0x11223344U);
code->Access(true, address, 0x55667788, RDX, RCX);
EXPECT_EQ(mmu.ReadForJit<u32>(address), 0x55667788U);
}
}
TEST_F(Jit64DCache, InhibitedBATBypassesCachedData)
{
mmu.WriteForJit<u32>(0x11111111, 0x80001000);
memory.Write_U32(0x22222222, 0x1000);
bats[0x80000000 >> PowerPC::BAT_INDEX_SHIFT] |= PowerPC::BAT_WI_BIT;
code->Access(false, 0x80001000, 0, RDX, R8);
EXPECT_EQ(code->result[R8], 0x22222222U);
code->Access(true, 0x80001000, 0x33333333, RDX, RCX);
EXPECT_EQ(memory.Read_U32(0x1000), 0x33333333U);
bats[0x80000000 >> PowerPC::BAT_INDEX_SHIFT] &= ~PowerPC::BAT_WI_BIT;
EXPECT_EQ(mmu.ReadForJit<u32>(0x80001000), 0x11111111U);
}
TEST_F(Jit64DCache, SharedRoutineStackAndDRFlag)
{
// Shared paired-load routines are emitted without a known block's feature flags and run
// with a return address on the stack. DR_ON is their explicit translation contract.
state.feature_flags = CPUEmuFeatureFlags{};
constexpr int flags = EmuCodeBlock::SAFE_LOADSTORE_NO_PROLOG |
EmuCodeBlock::SAFE_LOADSTORE_NO_UPDATE_PC |
EmuCodeBlock::SAFE_LOADSTORE_DR_ON;
memory.Write_U32(0x89abcdef, 0x1000);
for (int repeat = 0; repeat < 2; ++repeat)
{
code->Access(false, 0x80001000, 0, RDX, R8, 0, flags);
EXPECT_EQ(code->result[R8], 0x89abcdef);
code->Access(true, 0x80001000, 0x89abcdef, RDX, RCX, 0, flags);
EXPECT_EQ(mmu.ReadForJit<u32>(0x80001000), 0x89abcdef);
}
}
TEST_F(Jit64DCache, NativePLRUAndDirtyStateMatchMMU)
{
for (const u32 base : {0x80000000U, 0x90000000U})
{
state.dCache.Reset();
constexpr u32 set = 126;
for (u32 way = 0; way < PowerPC::CACHE_WAYS; ++way)
mmu.WriteForJit<u32>(0xffffffff, base + way * 4096 + set * 32);
for (u32 way = 0; way < PowerPC::CACHE_WAYS; ++way)
{
const u32 address = base + way * 4096 + set * 32;
for (const bool write : {false, true})
{
code->Access(write, address, 0xffffffff, RDX, R8);
for (u32 old_plru = 0; old_plru < 128; ++old_plru)
{
SCOPED_TRACE(testing::Message() << base << ' ' << way << ' ' << write << ' ' << old_plru);
auto& cache = state.dCache;
cache.plru[set] = static_cast<u8>(old_plru);
cache.modified[set] = 0x80;
if (write)
ASSERT_TRUE(mmu.TryWriteDCache32ForJit(0xffffffff, address));
else
ASSERT_EQ(mmu.TryReadDCache32ForJit(address), 0x100000000ULL);
const auto expected_plru = cache.plru[set];
const auto expected_dirty = cache.modified[set];
const auto expected_data = cache.data[set];
cache.plru[set] = static_cast<u8>(old_plru);
cache.modified[set] = 0x80;
code->run();
EXPECT_EQ(cache.plru[set], expected_plru);
EXPECT_EQ(cache.modified[set], expected_dirty);
EXPECT_EQ(cache.data[set], expected_data);
if (!write)
EXPECT_EQ(code->result[R8], 0xffffffff);
}
}
}
}
}
} // namespace