diff --git a/.gitignore b/.gitignore index 5af4dd6882..3df04761d9 100644 --- a/.gitignore +++ b/.gitignore @@ -66,3 +66,8 @@ CMakeUserPresets.json /capstone_local/ /capstone_runtime/ /capstone-*.whl + +# Local profiling helpers, captures, and diagnostic output +/.codex-tools/ +/.starlet-profile/ +/yaya48-starlet-log.txt diff --git a/Run-Wii-IOS-LLE.ps1 b/Run-Wii-IOS-LLE.ps1 index bbee81bcea..3abe4176d9 100644 --- a/Run-Wii-IOS-LLE.ps1 +++ b/Run-Wii-IOS-LLE.ps1 @@ -5,6 +5,8 @@ param( [switch]$DisableJIT, + [switch]$SafeTextureCache = $true, + [switch]$Wait ) @@ -16,7 +18,7 @@ if (-not (Test-Path -LiteralPath $dolphinPath)) { throw "DolphinNoGUI.exe is missing: $dolphinPath" } -$dolphin = Start-Process -FilePath $dolphinPath -ArgumentList @( +$dolphinArguments = @( '-u', $userPath, '-n', '0000000100000002', '-C', "Dolphin.Core.WiiStarletJIT=$(-not $DisableJIT)", @@ -27,7 +29,16 @@ $dolphin = Start-Process -FilePath $dolphinPath -ArgumentList @( '-C', 'Logger.Logs.IOS=True', '-v', 'D3D', '-p', 'win32' -) -WorkingDirectory $repoRoot -PassThru +) + +if ($SafeTextureCache) { + # Hash every texture byte: sparse samples can miss small CPU-rendered text updates. + # This is a per-run override and does not change the saved graphics configuration. + $dolphinArguments += @('-C', 'Graphics.Settings.SafeTextureCacheColorSamples=0') +} + +$dolphin = Start-Process -FilePath $dolphinPath -ArgumentList $dolphinArguments ` + -WorkingDirectory $repoRoot -PassThru $dolphin.PriorityClass = 'High' diff --git a/Source/Core/Core/HW/VideoInterface.cpp b/Source/Core/Core/HW/VideoInterface.cpp index 21d01a7435..91499ee116 100644 --- a/Source/Core/Core/HW/VideoInterface.cpp +++ b/Source/Core/Core/HW/VideoInterface.cpp @@ -317,7 +317,7 @@ void VideoInterfaceManager::RegisterMMIO(MMIO::Mapping* mmio, u32 base) mmio->Register( base | VI_VERTICAL_BEAM_POSITION, MMIO::ComplexRead([](Core::System& system, u32) { auto& vi = system.GetVideoInterface(); - return 1 + (vi.m_half_line_count) / 2; + return vi.GetVerticalBeamPosition(); }), MMIO::ComplexWrite([](Core::System& system, u32, u16 val) { auto& vi = system.GetVideoInterface(); diff --git a/Source/Core/Core/HW/VideoInterface.h b/Source/Core/Core/HW/VideoInterface.h index 6bf762aeab..a931f01c2d 100644 --- a/Source/Core/Core/HW/VideoInterface.h +++ b/Source/Core/Core/HW/VideoInterface.h @@ -389,6 +389,9 @@ public: u32 GetTicksPerHalfLine() const; u32 GetTicksPerField() const; + // Current one-based vertical beam position exposed by VI_VERTICAL_BEAM_POSITION. + u16 GetVerticalBeamPosition() const { return static_cast(1 + m_half_line_count / 2); } + // Not adjusted by VBI Clock Override. u32 GetNominalTicksPerHalfLine() const; diff --git a/Source/Core/Core/IOS/Starlet/ARMCore.cpp b/Source/Core/Core/IOS/Starlet/ARMCore.cpp index 706a7c6b7e..df96796cf9 100644 --- a/Source/Core/Core/IOS/Starlet/ARMCore.cpp +++ b/Source/Core/Core/IOS/Starlet/ARMCore.cpp @@ -7,6 +7,7 @@ #include #include #include +#include #include #include @@ -201,43 +202,95 @@ u32 ARMCore::TranslateVirtualAddress(u32 address) const return physical_address; }; + return cache_translation(WalkPageTables(modified_address, nullptr, nullptr, nullptr, nullptr)); +} + +u32 ARMCore::TranslateVirtualAddressForJit(u32 address, const u8** first_descriptor, + u32* first_descriptor_value, + const u8** second_descriptor, + u32* second_descriptor_value) const +{ + *first_descriptor = nullptr; + *first_descriptor_value = 0; + *second_descriptor = nullptr; + *second_descriptor_value = 0; + if ((m_cp15.control & CP15_CONTROL_MMU) == 0) + return address; + + const u32 modified_address = + address < 0x02000000 ? address | (m_cp15.process_id & 0xfe000000) : address; + const u32 physical_address = + WalkPageTables(modified_address, first_descriptor, first_descriptor_value, second_descriptor, + second_descriptor_value); + + // Keep the interpreter and generated data-access path coherent with the translation established + // for this native code block. + const u32 virtual_page = modified_address >> 10; + TLBEntry& entry = m_tlb[GetTLBIndex(virtual_page)]; + entry.virtual_page = virtual_page; + entry.physical_page = physical_address & ~0x3ffU; + entry.generation = m_tlb_generation; + return physical_address; +} + +u32 ARMCore::WalkPageTables(u32 modified_address, const u8** first_descriptor, + u32* first_descriptor_value, const u8** second_descriptor, + u32* second_descriptor_value) const +{ + const auto read_descriptor = [&](u32 physical_address, const u8** host_pointer, + u32* host_value) { + const u8* const pointer = m_bus.GetDirectMemoryPointer(physical_address, sizeof(u32)); + if (host_pointer) + *host_pointer = pointer; + if (host_value) + { + *host_value = 0; + if (pointer) + std::memcpy(host_value, pointer, sizeof(u32)); + } + return ReadPhysical32(physical_address); + }; + const u32 first_level_address = (m_cp15.translation_table_base & 0xffffc000) | ((modified_address >> 18) & 0x3ffc); - const u32 first_level = ReadPhysical32(first_level_address); + const u32 first_level = + read_descriptor(first_level_address, first_descriptor, first_descriptor_value); switch (first_level & 3) { case 1: // Coarse second-level table. { const u32 second_level_address = (first_level & 0xfffffc00) | ((modified_address >> 10) & 0x3fc); - const u32 second_level = ReadPhysical32(second_level_address); + const u32 second_level = + read_descriptor(second_level_address, second_descriptor, second_descriptor_value); switch (second_level & 3) { case 1: // 64 KiB large page. - return cache_translation((second_level & 0xffff0000) | (modified_address & 0xffff)); + return (second_level & 0xffff0000) | (modified_address & 0xffff); case 2: case 3: // 4 KiB small page; extended small pages share this mapping shape. - return cache_translation((second_level & 0xfffff000) | (modified_address & 0xfff)); + return (second_level & 0xfffff000) | (modified_address & 0xfff); default: - return cache_translation(modified_address); + return modified_address; } } case 2: // 1 MiB section. - return cache_translation((first_level & 0xfff00000) | (modified_address & 0x000fffff)); + return (first_level & 0xfff00000) | (modified_address & 0x000fffff); case 3: // Fine second-level table. { const u32 second_level_address = (first_level & 0xfffff000) | ((modified_address >> 8) & 0xffc); - const u32 second_level = ReadPhysical32(second_level_address); + const u32 second_level = + read_descriptor(second_level_address, second_descriptor, second_descriptor_value); switch (second_level & 3) { case 1: - return cache_translation((second_level & 0xffff0000) | (modified_address & 0xffff)); + return (second_level & 0xffff0000) | (modified_address & 0xffff); case 2: - return cache_translation((second_level & 0xfffff000) | (modified_address & 0xfff)); + return (second_level & 0xfffff000) | (modified_address & 0xfff); case 3: // 1 KiB tiny page. - return cache_translation((second_level & 0xfffffc00) | (modified_address & 0x3ff)); + return (second_level & 0xfffffc00) | (modified_address & 0x3ff); default: - return cache_translation(modified_address); + return modified_address; } } default: @@ -246,7 +299,7 @@ u32 ARMCore::TranslateVirtualAddress(u32 address) const // fallback keeps diagnostics observable meanwhile. Cache that provisional translation just // like a mapped page: real software must invalidate the TLB after changing a page-table entry, // and without this cache the JIT would side-exit forever on every identity access. - return cache_translation(modified_address); + return modified_address; } } @@ -864,6 +917,213 @@ size_t ARMCore::GetJitCompiledBlockCount() const #endif } +u64 ARMCore::GetJitLifetimeCompiledBlockCount() const +{ +#if defined(_M_X86_64) + return m_jit ? m_jit->GetLifetimeCompiledBlockCount() : 0; +#else + return 0; +#endif +} + +u64 ARMCore::GetJitBlockLookupCount() const +{ +#if defined(_M_X86_64) + return m_jit ? m_jit->GetBlockLookupCount() : 0; +#else + return 0; +#endif +} + +u64 ARMCore::GetJitBlockCacheHitCount() const +{ +#if defined(_M_X86_64) + return m_jit ? m_jit->GetBlockCacheHitCount() : 0; +#else + return 0; +#endif +} + +u64 ARMCore::GetJitDispatchSlowCount() const +{ +#if defined(_M_X86_64) + return m_jit ? m_jit->GetDispatchSlowCount() : 0; +#else + return 0; +#endif +} + +u64 ARMCore::GetJitDispatchStaleGenerationCount() const +{ +#if defined(_M_X86_64) + return m_jit ? m_jit->GetDispatchStaleGenerationCount() : 0; +#else + return 0; +#endif +} + +u64 ARMCore::GetJitDispatchPhysicalAliasCount() const +{ +#if defined(_M_X86_64) + return m_jit ? m_jit->GetDispatchPhysicalAliasCount() : 0; +#else + return 0; +#endif +} + +u64 ARMCore::GetJitDispatchCollisionCount() const +{ +#if defined(_M_X86_64) + return m_jit ? m_jit->GetDispatchCollisionCount() : 0; +#else + return 0; +#endif +} + +u64 ARMCore::GetJitDispatchKeyMissCount() const +{ +#if defined(_M_X86_64) + return m_jit ? m_jit->GetDispatchKeyMissCount() : 0; +#else + return 0; +#endif +} + +u64 ARMCore::GetJitDispatchPageMissCount() const +{ +#if defined(_M_X86_64) + return m_jit ? m_jit->GetDispatchPageMissCount() : 0; +#else + return 0; +#endif +} + +u64 ARMCore::GetJitDispatchEmptyEntryCount() const +{ +#if defined(_M_X86_64) + return m_jit ? m_jit->GetDispatchEmptyEntryCount() : 0; +#else + return 0; +#endif +} + +u64 ARMCore::GetJitDispatchGeneratedKeyMismatchCount() const +{ +#if defined(_M_X86_64) + return m_jit ? m_jit->GetDispatchGeneratedKeyMismatchCount() : 0; +#else + return 0; +#endif +} + +u64 ARMCore::GetJitDispatchGeneratedSetMismatchCount() const +{ +#if defined(_M_X86_64) + return m_jit ? m_jit->GetDispatchGeneratedSetMismatchCount() : 0; +#else + return 0; +#endif +} + +u64 ARMCore::GetJitDispatchKeyMissWithExactMatchCount() const +{ +#if defined(_M_X86_64) + return m_jit ? m_jit->GetDispatchKeyMissWithExactMatchCount() : 0; +#else + return 0; +#endif +} + +u64 ARMCore::GetJitDispatchKeyMissWithPhysicalAliasCount() const +{ +#if defined(_M_X86_64) + return m_jit ? m_jit->GetDispatchKeyMissWithPhysicalAliasCount() : 0; +#else + return 0; +#endif +} + +u64 ARMCore::GetJitDispatchKeyMissWithSamePhysicalPageCount() const +{ +#if defined(_M_X86_64) + return m_jit ? m_jit->GetDispatchKeyMissWithSamePhysicalPageCount() : 0; +#else + return 0; +#endif +} + +u64 ARMCore::GetJitDispatchKeyMissWithEmptySlotCount() const +{ +#if defined(_M_X86_64) + return m_jit ? m_jit->GetDispatchKeyMissWithEmptySlotCount() : 0; +#else + return 0; +#endif +} + +u64 ARMCore::GetJitDispatchKeyMissWithFullSetCount() const +{ +#if defined(_M_X86_64) + return m_jit ? m_jit->GetDispatchKeyMissWithFullSetCount() : 0; +#else + return 0; +#endif +} + +u64 ARMCore::GetJitDispatchKeyMissWithStaleTranslationCount() const +{ +#if defined(_M_X86_64) + return m_jit ? m_jit->GetDispatchKeyMissWithStaleTranslationCount() : 0; +#else + return 0; +#endif +} + +u64 ARMCore::GetJitFastEntryEmptyInsertCount() const +{ +#if defined(_M_X86_64) + return m_jit ? m_jit->GetFastEntryEmptyInsertCount() : 0; +#else + return 0; +#endif +} + +u64 ARMCore::GetJitFastEntryReplacementCount() const +{ +#if defined(_M_X86_64) + return m_jit ? m_jit->GetFastEntryReplacementCount() : 0; +#else + return 0; +#endif +} + +u64 ARMCore::GetJitInstructionCacheClearCount() const +{ +#if defined(_M_X86_64) + return m_jit ? m_jit->GetInstructionCacheClearCount() : 0; +#else + return 0; +#endif +} + +u64 ARMCore::GetJitCodeSpaceClearCount() const +{ +#if defined(_M_X86_64) + return m_jit ? m_jit->GetCodeSpaceClearCount() : 0; +#else + return 0; +#endif +} + +u64 ARMCore::GetJitDiscardedBlockCount() const +{ +#if defined(_M_X86_64) + return m_jit ? m_jit->GetDiscardedBlockCount() : 0; +#else + return 0; +#endif +} + u64 ARMCore::GetJitAddressTranslationCount() const { #if defined(_M_X86_64) @@ -990,6 +1250,16 @@ u64 ARMCore::GetJitSlowSRAMPageWriteAccessCount(size_t page) const #endif } +std::vector> ARMCore::TakeHotJitSlowMemorySamples(size_t maximum_count) +{ +#if defined(_M_X86_64) + return m_jit ? m_jit->TakeHotSlowMemorySamples(maximum_count) : + std::vector>{}; +#else + return {}; +#endif +} + void ARMCore::RecordJitFallback(u32 address, bool thumb) { ++m_jit_fallback_instruction_count; diff --git a/Source/Core/Core/IOS/Starlet/ARMCore.h b/Source/Core/Core/IOS/Starlet/ARMCore.h index da402fa516..470e9ed7bc 100644 --- a/Source/Core/Core/IOS/Starlet/ARMCore.h +++ b/Source/Core/Core/IOS/Starlet/ARMCore.h @@ -64,6 +64,10 @@ public: virtual u8* GetFastmemSRAMBase() const { return nullptr; } virtual const bool* GetFastmemBoot0Mapped() const { return nullptr; } virtual const bool* GetFastmemSRAMSplitMode() const { return nullptr; } + // Returns a stable host pointer for ordinary physical memory. The pointer is used only to + // validate ARM page-table descriptors after guest TLB maintenance; MMIO and overlaid ROM must + // return nullptr so their observable reads continue through the bus. + virtual const u8* GetDirectMemoryPointer(u32 address, u32 size) const { return nullptr; } }; class ARMCore final @@ -164,6 +168,29 @@ public: u64 GetJitExecutedInstructions() const; u64 GetJitNativeExecutedInstructions() const; size_t GetJitCompiledBlockCount() const; + u64 GetJitLifetimeCompiledBlockCount() const; + u64 GetJitBlockLookupCount() const; + u64 GetJitBlockCacheHitCount() const; + u64 GetJitDispatchSlowCount() const; + u64 GetJitDispatchStaleGenerationCount() const; + u64 GetJitDispatchPhysicalAliasCount() const; + u64 GetJitDispatchCollisionCount() const; + u64 GetJitDispatchKeyMissCount() const; + u64 GetJitDispatchPageMissCount() const; + u64 GetJitDispatchEmptyEntryCount() const; + u64 GetJitDispatchGeneratedKeyMismatchCount() const; + u64 GetJitDispatchGeneratedSetMismatchCount() const; + u64 GetJitDispatchKeyMissWithExactMatchCount() const; + u64 GetJitDispatchKeyMissWithPhysicalAliasCount() const; + u64 GetJitDispatchKeyMissWithSamePhysicalPageCount() const; + u64 GetJitDispatchKeyMissWithEmptySlotCount() const; + u64 GetJitDispatchKeyMissWithFullSetCount() const; + u64 GetJitDispatchKeyMissWithStaleTranslationCount() const; + u64 GetJitFastEntryEmptyInsertCount() const; + u64 GetJitFastEntryReplacementCount() const; + u64 GetJitInstructionCacheClearCount() const; + u64 GetJitCodeSpaceClearCount() const; + u64 GetJitDiscardedBlockCount() const; u64 GetJitAddressTranslationCount() const; u64 GetJitSlowReadCount() const; u64 GetJitSlowWriteCount() const; @@ -178,6 +205,7 @@ public: u64 GetJitSlowSRAMPageAccessCount(size_t page) const; u64 GetJitSlowSRAMPageReadAccessCount(size_t page) const; u64 GetJitSlowSRAMPageWriteAccessCount(size_t page) const; + std::vector> TakeHotJitSlowMemorySamples(size_t maximum_count); u64 GetJitFallbackInstructionCount() const { return m_jit_fallback_instruction_count; } u64 GetMemoryPollEntryCount() const { return m_memory_poll_entry_count; } std::vector GetHotPCSamples(size_t maximum_count); @@ -246,6 +274,12 @@ private: void Write32(u32 address, u32 value); void WriteByte(u32 address, u8 value); u32 TranslateVirtualAddress(u32 address) const; + u32 TranslateVirtualAddressForJit(u32 address, const u8** first_descriptor, + u32* first_descriptor_value, const u8** second_descriptor, + u32* second_descriptor_value) const; + u32 WalkPageTables(u32 modified_address, const u8** first_descriptor, + u32* first_descriptor_value, const u8** second_descriptor, + u32* second_descriptor_value) const; u32 ReadPhysical32(u32 address) const; void InvalidateTLB(); u16 FetchThumbInstruction(u32 address); diff --git a/Source/Core/Core/IOS/Starlet/ARMJitX64.cpp b/Source/Core/Core/IOS/Starlet/ARMJitX64.cpp index e068d12885..381d840a11 100644 --- a/Source/Core/Core/IOS/Starlet/ARMJitX64.cpp +++ b/Source/Core/Core/IOS/Starlet/ARMJitX64.cpp @@ -76,6 +76,10 @@ ARMJitX64::ARMJitX64(ARMCore& core) : m_core(core) m_executed_instructions_offset = static_cast(reinterpret_cast(&m_core.m_executed_instructions) - base); m_control_offset = static_cast(reinterpret_cast(&m_core.m_cp15.control) - base); + m_translation_table_base_offset = static_cast( + reinterpret_cast(&m_core.m_cp15.translation_table_base) - base); + m_domain_access_control_offset = static_cast( + reinterpret_cast(&m_core.m_cp15.domain_access_control) - base); m_process_id_offset = static_cast(reinterpret_cast(&m_core.m_cp15.process_id) - base); m_tlb_generation_offset = @@ -99,6 +103,18 @@ void ARMJitX64::PoisonMemory() } void ARMJitX64::Clear() +{ + ++m_instruction_cache_clear_count; + RequestClear(); +} + +void ARMJitX64::ClearForCodeSpace() +{ + ++m_code_space_clear_count; + RequestClear(); +} + +void ARMJitX64::RequestClear() { // CP15 I-cache maintenance can be reached through a fallback at the end of the currently // executing host block. Defer destruction until its RET has brought us back to C++. @@ -107,8 +123,15 @@ void ARMJitX64::Clear() m_clear_pending = true; return; } + ClearCodeCache(); +} + +void ARMJitX64::ClearCodeCache() +{ + m_discarded_block_count += m_blocks.size(); m_blocks.clear(); std::ranges::fill(m_fast_entries, FastEntry{}); + std::ranges::fill(m_fast_entry_next_victim, 0); SetCodePtr(m_block_code_begin, region + region_size); } @@ -117,13 +140,12 @@ void ARMJitX64::InvalidateTranslationContext() // A TLB invalidation changes which physical page a virtual PC resolves to, but it does not // invalidate the ARM instruction cache. Keep already translated physical code and force the // generated dispatcher to resolve the next virtual PC through the current page tables. Every - // fast entry carries the ARMCore TLB generation, so the common invalidation is O(1). This is - // critical for IOS, which flushes its TLB on virtually every process switch; clearing the whole - // 65,536-entry array here previously consumed most of the host CPU. Generation zero is skipped by - // ARMCore. If the 32-bit counter eventually wraps back to one, clear ancient generation-one - // entries once to prevent an alias after the wrap. + // shared page translation carries the ARMCore TLB generation, so the common invalidation is + // O(1). This is critical for IOS, which flushes its TLB on virtually every process switch. + // Generation zero is skipped by ARMCore. If the 32-bit counter eventually wraps back to one, + // clear ancient generation-one page translations once to prevent a generation alias. if (m_core.m_tlb_generation == 1) - std::ranges::fill(m_fast_entries, FastEntry{}); + std::ranges::fill(m_fast_translations, FastTranslationEntry{}); } u32 ARMJitX64::Run(u64 cycle_budget) @@ -138,7 +160,7 @@ u32 ARMJitX64::Run(u64 cycle_budget) if (m_clear_pending) { m_clear_pending = false; - Clear(); + ClearCodeCache(); } const u32 executed = static_cast(result); m_executed_instructions += executed; @@ -168,9 +190,11 @@ void ARMJitX64::GenerateDispatcher() CMP(8, MatR(RAX), Imm8(0)); FixupBranch invalidated = J_CC(CC_NE, Jump::Near); - // Direct-mapped native block cache. The key includes CPSR.T in bit zero and uses the ARM926 - // modified virtual address (MVA) for low FCSE addresses. Different IOS process identifiers can - // therefore retain independent hot entries without invalidating the cache on every c13 write. + // Four-way native block cache. The key includes CPSR.T in bit zero and uses the ARM926 modified + // virtual address (MVA) for low FCSE addresses. Different IOS process identifiers can therefore + // retain independent hot entries without invalidating the cache on every c13 write. Four ways + // are important here: IOS regularly alternates among several hot basic blocks whose MVAs used + // to evict each other millions of times in the old direct-mapped cache. MOV(32, R(EAX), MRegister(15)); MOV(32, R(ECX), MCPSR()); SHR(32, R(ECX), Imm8(5)); @@ -187,30 +211,181 @@ void ARMJitX64::GenerateDispatcher() SetJumpTarget(mmu_disabled); SetJumpTarget(outside_fcse); - // Fold the FCSE PID bits down into the cache index. A plain low-bit mask makes every process - // collide because PID occupies MVA[31:25]. The full MVA remains in FastEntry::key for safety. + // Multiplicative hashing folds both FCSE PID and low instruction-address bits into the set. Each + // set occupies one 64-byte host cache line (four 16-byte FastEntry values). MOV(32, R(EDX), R(EAX)); - SHR(32, R(EDX), Imm8(1)); - MOV(32, R(ECX), R(EAX)); - SHR(32, R(ECX), Imm8(17)); - XOR(32, R(EDX), R(ECX)); - AND(32, R(EDX), Imm32(static_cast(FAST_ENTRY_COUNT - 1))); - SHL(64, R(RDX), Imm8(4)); + IMUL(32, EDX, R(EDX), Imm32(0x9e3779b1U)); + SHR(32, R(EDX), Imm8(16)); + SHL(64, R(RDX), Imm8(6)); MOV(64, R(R11), ImmPtr(m_fast_entries.data())); ADD(64, R(R11), R(RDX)); - MOV(32, R(R8), MDisp(JIT_CORE, m_tlb_generation_offset)); - CMP(32, MDisp(R11, static_cast(offsetof(FastEntry, tlb_generation))), R(R8)); - FixupBranch stale_generation = J_CC(CC_NE); + + // With the MMU disabled the virtual and physical identities are identical, so a key-only lookup + // is sufficient and TLB maintenance is irrelevant. + MOV(32, R(ECX), MDisp(JIT_CORE, m_control_offset)); + TEST(32, R(ECX), Imm32(1)); + FixupBranch mmu_fast_translation = J_CC(CC_NE, Jump::Near); + MOV(32, R(R8), R(EAX)); + AND(32, R(R8), Imm32(~0x3ffU)); CMP(32, MDisp(R11, static_cast(offsetof(FastEntry, key))), R(EAX)); - FixupBranch cache_miss = J_CC(CC_NE); + FixupBranch no_mmu_way0_key_miss = J_CC(CC_NE, Jump::Near); + CMP(32, MDisp(R11, static_cast(offsetof(FastEntry, physical_page))), R(R8)); + FixupBranch no_mmu_hit_way0 = J_CC(CC_E, Jump::Near); + SetJumpTarget(no_mmu_way0_key_miss); + CMP(32, MDisp(R11, static_cast(sizeof(FastEntry) + offsetof(FastEntry, key))), R(EAX)); + FixupBranch no_mmu_way1_key_miss = J_CC(CC_NE, Jump::Near); + CMP(32, MDisp(R11, static_cast(sizeof(FastEntry) + offsetof(FastEntry, physical_page))), + R(R8)); + FixupBranch no_mmu_hit_way1 = J_CC(CC_E, Jump::Near); + SetJumpTarget(no_mmu_way1_key_miss); + CMP(32, MDisp(R11, static_cast(2 * sizeof(FastEntry) + offsetof(FastEntry, key))), R(EAX)); + FixupBranch no_mmu_way2_key_miss = J_CC(CC_NE, Jump::Near); + CMP(32, MDisp(R11, + static_cast(2 * sizeof(FastEntry) + offsetof(FastEntry, physical_page))), + R(R8)); + FixupBranch no_mmu_hit_way2 = J_CC(CC_E, Jump::Near); + SetJumpTarget(no_mmu_way2_key_miss); + CMP(32, MDisp(R11, static_cast(3 * sizeof(FastEntry) + offsetof(FastEntry, key))), R(EAX)); + FixupBranch no_mmu_cache_miss_key = J_CC(CC_NE, Jump::Near); + CMP(32, MDisp(R11, + static_cast(3 * sizeof(FastEntry) + offsetof(FastEntry, physical_page))), + R(R8)); + FixupBranch no_mmu_cache_miss_page = J_CC(CC_NE, Jump::Near); + ADD(64, R(R11), Imm8(static_cast(3 * sizeof(FastEntry)))); + FixupBranch no_mmu_way_selected = J(Jump::Near); + SetJumpTarget(no_mmu_hit_way2); + ADD(64, R(R11), Imm8(static_cast(2 * sizeof(FastEntry)))); + FixupBranch no_mmu_way2_selected = J(Jump::Near); + SetJumpTarget(no_mmu_hit_way1); + ADD(64, R(R11), Imm8(static_cast(sizeof(FastEntry)))); + SetJumpTarget(no_mmu_hit_way0); + SetJumpTarget(no_mmu_way2_selected); + SetJumpTarget(no_mmu_way_selected); + MOV(64, R(R10), MDisp(R11, static_cast(offsetof(FastEntry, entry)))); + TEST(64, R(R10), R(R10)); + FixupBranch empty_entry_no_mmu = J_CC(CC_Z, Jump::Near); + JMPptr(R(R10)); + SetJumpTarget(mmu_fast_translation); + + // ARM926 TLB maintenance invalidates translations by page. Validate one shared 1 KiB physical + // identity per page instead of sending every decoded block through C++ after each IOS context + // switch. A true page remap still rejects the old native block below. + MOV(32, R(R9), R(EAX)); + SHR(32, R(R9), Imm8(10)); + MOV(32, R(R8), R(R9)); + MOV(32, R(ECX), R(R9)); + SHR(32, R(ECX), Imm8(12)); + XOR(32, R(R8), R(ECX)); + AND(32, R(R8), Imm32(static_cast(FAST_TRANSLATION_ENTRY_COUNT - 1))); + SHL(64, R(R8), Imm8(6)); + MOV(64, R(R10), ImmPtr(m_fast_translations.data())); + ADD(64, R(R10), R(R8)); + + CMP(32, MDisp(R10, static_cast(offsetof(FastTranslationEntry, virtual_page))), R(R9)); + FixupBranch stale_page = J_CC(CC_NE, Jump::Near); + MOV(32, R(R8), MDisp(JIT_CORE, m_tlb_generation_offset)); + CMP(32, MDisp(R10, static_cast(offsetof(FastTranslationEntry, tlb_generation))), R(R8)); + FixupBranch current_translation = J_CC(CC_E, Jump::Near); + + // IOS invalidates its ARM926 TLB on almost every process switch, even when the page tables are + // unchanged. Revalidate the exact descriptors that established this translation in generated + // code. A changed TTBR or descriptor still takes the complete C++ page-table walk below, while + // unchanged mappings avoid millions of dispatcher side exits. + MOV(32, R(ECX), MDisp(JIT_CORE, m_translation_table_base_offset)); + CMP(32, MDisp(R10, static_cast(offsetof(FastTranslationEntry, translation_table_base))), + R(ECX)); + FixupBranch stale_translation_table = J_CC(CC_NE, Jump::Near); + MOV(64, R(R8), MDisp(R10, static_cast(offsetof(FastTranslationEntry, first_descriptor)))); + TEST(64, R(R8), R(R8)); + FixupBranch unavailable_first_descriptor = J_CC(CC_Z, Jump::Near); + MOV(32, R(ECX), MatR(R8)); + CMP(32, R(ECX), + MDisp(R10, static_cast(offsetof(FastTranslationEntry, first_descriptor_value)))); + FixupBranch changed_first_descriptor = J_CC(CC_NE, Jump::Near); + MOV(64, R(R8), MDisp(R10, static_cast(offsetof(FastTranslationEntry, second_descriptor)))); + TEST(64, R(R8), R(R8)); + FixupBranch no_second_descriptor = J_CC(CC_Z, Jump::Near); + MOV(32, R(ECX), MatR(R8)); + CMP(32, R(ECX), + MDisp(R10, static_cast(offsetof(FastTranslationEntry, second_descriptor_value)))); + FixupBranch changed_second_descriptor = J_CC(CC_NE, Jump::Near); + SetJumpTarget(no_second_descriptor); + MOV(32, R(R8), MDisp(JIT_CORE, m_tlb_generation_offset)); + MOV(32, MDisp(R10, static_cast(offsetof(FastTranslationEntry, tlb_generation))), R(R8)); + SetJumpTarget(current_translation); + MOV(32, R(R8), MDisp(R10, static_cast(offsetof(FastTranslationEntry, physical_page)))); + + // The same MVA can legitimately resolve to different physical pages in different IOS address + // spaces. Select a way only when both identities match, allowing the four physical mappings to + // coexist instead of overwriting one another on every process switch. + CMP(32, MDisp(R11, static_cast(offsetof(FastEntry, key))), R(EAX)); + FixupBranch mmu_way0_key_miss = J_CC(CC_NE, Jump::Near); + CMP(32, MDisp(R11, static_cast(offsetof(FastEntry, physical_page))), R(R8)); + FixupBranch mmu_hit_way0 = J_CC(CC_E, Jump::Near); + SetJumpTarget(mmu_way0_key_miss); + CMP(32, MDisp(R11, static_cast(sizeof(FastEntry) + offsetof(FastEntry, key))), R(EAX)); + FixupBranch mmu_way1_key_miss = J_CC(CC_NE, Jump::Near); + CMP(32, MDisp(R11, static_cast(sizeof(FastEntry) + offsetof(FastEntry, physical_page))), + R(R8)); + FixupBranch mmu_hit_way1 = J_CC(CC_E, Jump::Near); + SetJumpTarget(mmu_way1_key_miss); + CMP(32, MDisp(R11, static_cast(2 * sizeof(FastEntry) + offsetof(FastEntry, key))), R(EAX)); + FixupBranch mmu_way2_key_miss = J_CC(CC_NE, Jump::Near); + CMP(32, MDisp(R11, + static_cast(2 * sizeof(FastEntry) + offsetof(FastEntry, physical_page))), + R(R8)); + FixupBranch mmu_hit_way2 = J_CC(CC_E, Jump::Near); + SetJumpTarget(mmu_way2_key_miss); + CMP(32, MDisp(R11, static_cast(3 * sizeof(FastEntry) + offsetof(FastEntry, key))), R(EAX)); + FixupBranch mmu_cache_miss_key = J_CC(CC_NE, Jump::Near); + CMP(32, MDisp(R11, + static_cast(3 * sizeof(FastEntry) + offsetof(FastEntry, physical_page))), + R(R8)); + FixupBranch mmu_cache_miss_page = J_CC(CC_NE, Jump::Near); + ADD(64, R(R11), Imm8(static_cast(3 * sizeof(FastEntry)))); + FixupBranch mmu_way_selected = J(Jump::Near); + SetJumpTarget(mmu_hit_way2); + ADD(64, R(R11), Imm8(static_cast(2 * sizeof(FastEntry)))); + FixupBranch mmu_way2_selected = J(Jump::Near); + SetJumpTarget(mmu_hit_way1); + ADD(64, R(R11), Imm8(static_cast(sizeof(FastEntry)))); + SetJumpTarget(mmu_hit_way0); + SetJumpTarget(mmu_way2_selected); + SetJumpTarget(mmu_way_selected); MOV(64, R(R11), MDisp(R11, static_cast(offsetof(FastEntry, entry)))); TEST(64, R(R11), R(R11)); FixupBranch empty_entry = J_CC(CC_Z); JMPptr(R(R11)); - SetJumpTarget(stale_generation); - SetJumpTarget(cache_miss); + SetJumpTarget(stale_page); + SetJumpTarget(stale_translation_table); + SetJumpTarget(unavailable_first_descriptor); + SetJumpTarget(changed_first_descriptor); + SetJumpTarget(changed_second_descriptor); + MOV(32, R(ABI_PARAM4), R(EDX)); + MOV(32, R(ABI_PARAM3), R(EAX)); + MOV(32, R(ABI_PARAM2), Imm32(static_cast(DispatchReason::Translation))); + FixupBranch dispatch_reason_translation = J(Jump::Near); + SetJumpTarget(no_mmu_cache_miss_key); + SetJumpTarget(mmu_cache_miss_key); + MOV(32, R(ABI_PARAM4), R(EDX)); + MOV(32, R(ABI_PARAM3), R(EAX)); + MOV(32, R(ABI_PARAM2), Imm32(static_cast(DispatchReason::KeyMiss))); + FixupBranch dispatch_reason_key = J(Jump::Near); + SetJumpTarget(no_mmu_cache_miss_page); + SetJumpTarget(mmu_cache_miss_page); + MOV(32, R(ABI_PARAM4), R(EDX)); + MOV(32, R(ABI_PARAM3), R(EAX)); + MOV(32, R(ABI_PARAM2), Imm32(static_cast(DispatchReason::PageMiss))); + FixupBranch dispatch_reason_page = J(Jump::Near); SetJumpTarget(empty_entry); + SetJumpTarget(empty_entry_no_mmu); + MOV(32, R(ABI_PARAM4), R(EDX)); + MOV(32, R(ABI_PARAM3), R(EAX)); + MOV(32, R(ABI_PARAM2), Imm32(static_cast(DispatchReason::EmptyEntry))); + SetJumpTarget(dispatch_reason_translation); + SetJumpTarget(dispatch_reason_key); + SetJumpTarget(dispatch_reason_page); MOV(64, R(ABI_PARAM1), ImmPtr(this)); ABI_CallFunction(Dispatch); TEST(64, R(RAX), R(RAX)); @@ -245,48 +420,154 @@ u32 ARMJitX64::MakeFastEntryKey(u32 address, u32 control, u32 process_id) return address; } -size_t ARMJitX64::GetFastEntryIndex(u32 key) +size_t ARMJitX64::GetFastEntrySetIndex(u32 key) { - return ((key >> 1) ^ (key >> 17)) & (FAST_ENTRY_COUNT - 1); + return static_cast(key * 0x9e3779b1U) >> 16; } -const u8* ARMJitX64::Dispatch(ARMJitX64* jit) +const u8* ARMJitX64::Dispatch(ARMJitX64* jit, DispatchReason reason, u32 generated_key, + u32 generated_set_offset) { const bool thumb = (jit->m_core.m_cpsr & ARMCore::CPSR_T) != 0; const u32 key = jit->m_core.m_registers[15] | (thumb ? 1U : 0U); - Block* const block = jit->GetOrCompileBlock(key); - if (!block || !block->runnable) - return nullptr; - const u32 fast_key = MakeFastEntryKey(key, jit->m_core.m_cp15.control, jit->m_core.m_cp15.process_id); - FastEntry& fast_entry = jit->m_fast_entries[GetFastEntryIndex(fast_key)]; - fast_entry.key = fast_key; - fast_entry.tlb_generation = jit->m_core.m_tlb_generation; - fast_entry.entry = block->entry; + const u32 virtual_page = fast_key >> 10; + FastTranslationEntry& translation = + jit->m_fast_translations[(virtual_page ^ (virtual_page >> 12)) & + (FAST_TRANSLATION_ENTRY_COUNT - 1)]; + const bool mmu_enabled = (jit->m_core.m_cp15.control & 1) != 0; + const bool translation_is_current = + !mmu_enabled || (translation.tlb_generation == jit->m_core.m_tlb_generation && + translation.virtual_page == virtual_page); + const size_t set_index = GetFastEntrySetIndex(fast_key); + const size_t set_base = set_index * FAST_ENTRY_WAYS; + if (generated_key != fast_key) + ++jit->m_dispatch_generated_key_mismatch_count; + if (generated_set_offset != set_index * FAST_ENTRY_WAYS * sizeof(FastEntry)) + ++jit->m_dispatch_generated_set_mismatch_count; + ++jit->m_dispatch_slow_count; + if (reason == DispatchReason::Translation) + ++jit->m_dispatch_stale_generation_count; + else if (reason == DispatchReason::KeyMiss) + ++jit->m_dispatch_key_miss_count; + else if (reason == DispatchReason::PageMiss) + ++jit->m_dispatch_page_miss_count; + else if (reason == DispatchReason::EmptyEntry) + ++jit->m_dispatch_empty_entry_count; + + u32 physical_address = 0; + const u8* first_descriptor = nullptr; + const u8* second_descriptor = nullptr; + u32 first_descriptor_value = 0; + u32 second_descriptor_value = 0; + Block* const block = + jit->GetOrCompileBlock(key, &physical_address, &first_descriptor, &first_descriptor_value, + &second_descriptor, &second_descriptor_value); + if (!block || !block->runnable) + return nullptr; + const u32 physical_page = physical_address & ~0x3ffU; + + FastEntry* matching_entry = nullptr; + FastEntry* empty_entry = nullptr; + bool has_other_physical_mapping = false; + bool has_same_physical_page = false; + for (size_t way = 0; way < FAST_ENTRY_WAYS; ++way) + { + FastEntry& entry = jit->m_fast_entries[set_base + way]; + if (entry.key == fast_key && entry.physical_page == physical_page) + { + matching_entry = &entry; + break; + } + if (entry.key == fast_key && entry.entry != nullptr) + has_other_physical_mapping = true; + if (entry.physical_page == physical_page && entry.entry != nullptr) + has_same_physical_page = true; + if (entry.entry == nullptr && empty_entry == nullptr) + empty_entry = &entry; + } + if (translation_is_current && matching_entry == nullptr) + { + if (has_other_physical_mapping) + ++jit->m_dispatch_physical_alias_count; + else if (empty_entry == nullptr) + ++jit->m_dispatch_collision_count; + } + if (reason == DispatchReason::KeyMiss) + { + if (matching_entry != nullptr) + ++jit->m_dispatch_key_miss_with_exact_match_count; + if (has_other_physical_mapping) + ++jit->m_dispatch_key_miss_with_physical_alias_count; + if (has_same_physical_page) + ++jit->m_dispatch_key_miss_with_same_physical_page_count; + if (empty_entry != nullptr) + ++jit->m_dispatch_key_miss_with_empty_slot_count; + else + ++jit->m_dispatch_key_miss_with_full_set_count; + if (!translation_is_current) + ++jit->m_dispatch_key_miss_with_stale_translation_count; + } + + translation.virtual_page = virtual_page; + translation.physical_page = physical_page; + translation.tlb_generation = jit->m_core.m_tlb_generation; + translation.translation_table_base = jit->m_core.m_cp15.translation_table_base; + translation.first_descriptor = first_descriptor; + translation.second_descriptor = second_descriptor; + translation.first_descriptor_value = first_descriptor_value; + translation.second_descriptor_value = second_descriptor_value; + FastEntry* fast_entry = matching_entry; + if (fast_entry == nullptr) + { + if (empty_entry != nullptr) + { + fast_entry = empty_entry; + ++jit->m_fast_entry_empty_insert_count; + } + else + { + u8& next_victim = jit->m_fast_entry_next_victim[set_index]; + fast_entry = &jit->m_fast_entries[set_base + next_victim]; + next_victim = static_cast((next_victim + 1) & (FAST_ENTRY_WAYS - 1)); + ++jit->m_fast_entry_replacement_count; + } + } + fast_entry->key = fast_key; + fast_entry->physical_page = physical_page; + fast_entry->entry = block->entry; return block->entry; } -ARMJitX64::Block* ARMJitX64::GetOrCompileBlock(u32 address) +ARMJitX64::Block* ARMJitX64::GetOrCompileBlock(u32 address, u32* physical_address_out, + const u8** first_descriptor_out, + u32* first_descriptor_value_out, + const u8** second_descriptor_out, + u32* second_descriptor_value_out) { + ++m_block_lookup_count; const bool thumb = (address & 1) != 0; const u32 virtual_address = address & ~1U; - const u32 physical_address = m_core.TranslateVirtualAddress(virtual_address); + const u32 physical_address = m_core.TranslateVirtualAddressForJit( + virtual_address, first_descriptor_out, first_descriptor_value_out, second_descriptor_out, + second_descriptor_value_out); + *physical_address_out = physical_address; const u64 block_key = (static_cast(physical_address) << 32) | address; if (const auto it = m_blocks.find(block_key); it != m_blocks.end()) + { + ++m_block_cache_hit_count; return &it->second; + } if (IsAlmostFull()) { - if (m_is_running) - { - m_clear_pending = true; - return nullptr; - } - Clear(); + ClearForCodeSpace(); + return nullptr; } Block block = CompileBlock(virtual_address, thumb); + ++m_lifetime_compiled_block_count; return &m_blocks.emplace(block_key, block).first->second; } @@ -469,11 +750,60 @@ ARMJitX64::Block ARMJitX64::CompileBlock(u32 address, bool thumb) return {.entry = entry, .instruction_count = instruction_count, .native_instruction_count = native_instruction_count, - .runnable = native_instruction_count != 0}; + // A fallback-only block is still translated host code: it calls the exact interpreter + // helper, accounts the guest instruction, and returns through the native dispatcher. + // Rejecting it here made every unsupported hot instruction repeat address translation, + // unordered-map lookup and ABI dispatch in C++ before ARMCore interpreted it anyway. + // Caching the already-emitted wrapper preserves identical instruction semantics while + // removing that redundant lookup path. + .runnable = true}; } bool ARMJitX64::EmitDirectARM(u32 instruction, u32 address, bool* terminal, bool* dispatcher_exit) { + // The IOS scheduler executes a small set of ARM926 system-control writes in virtually every + // syscall and interrupt path. Decode those architectural operations once while compiling the + // block instead of entering the complete interpreter decoder every time the block runs. + // + // Keep TLB maintenance terminal: software may have changed a page table immediately before the + // MCR, so compiling subsequent guest instructions through the pre-invalidation mapping would be + // incorrect. The helper performs the same generation change as ARMCore::WriteCP15 and resumes at + // the following ARM instruction through the generated dispatcher. + const bool is_mcr_p15 = (instruction & 0x0f100f10) == 0x0e000f10; + if ((instruction >> 28) == 0xe && is_mcr_p15) + { + const u32 opcode1 = (instruction >> 21) & 7; + const u32 crn = (instruction >> 16) & 0xf; + const u32 rd = (instruction >> 12) & 0xf; + const u32 crm = instruction & 0xf; + const u32 opcode2 = (instruction >> 5) & 7; + + if (opcode1 == 0 && crn == 3 && rd != 15) + { + MOV(32, R(EAX), MRegister(rd)); + MOV(32, MDomainAccessControl(), R(EAX)); + return true; + } + + // ARMCore currently models only WFI and instruction-cache-affecting c7 operations. All other + // c7 writes are architecturally harmless in our single-host-thread memory model. In + // particular IOS's c7,c6,1 and c7,c10,1 forms are among its hottest privileged instructions. + const bool is_wfi = crn == 7 && crm == 0 && opcode2 == 4; + const bool affects_instruction_cache = crn == 7 && (crm == 5 || crm == 7); + if (opcode1 == 0 && crn == 7 && !is_wfi && !affects_instruction_cache) + return true; + + if (opcode1 == 0 && crn == 8) + { + FlushRegisterCache(); + MOV(64, R(ABI_PARAM1), ImmPtr(this)); + MOV(32, R(ABI_PARAM2), Imm32(address)); + ABI_CallFunction(InvalidateTLB); + *terminal = true; + return true; + } + } + // MCR p15, 0, Rd, c7, c10, 4 is ARM926 Drain Write Buffer. It is an ordering barrier, not an // instruction-cache invalidation, and has no additional observable work in this single-host- // thread memory model. IOS executes it in hot synchronization paths. @@ -640,6 +970,31 @@ bool ARMJitX64::EmitDirectARM(u32 instruction, u32 address, bool* terminal, bool return true; } + // IOS's crypto/bootstrap code contains tight UMULL loops. A single such instruction accounted + // for more than a million interpreter fallbacks during the first seconds of a real NAND boot. + // Translate the complete ARMv5 long-multiply family here; this keeps the arithmetic and optional + // NZ update architectural while eliminating the generic decoder/dispatcher round trip. + if ((instruction & 0x0f8000f0) == 0x00800090) + { + const u32 rd_hi = (instruction >> 16) & 0xf; + const u32 rd_lo = (instruction >> 12) & 0xf; + const u32 rs = (instruction >> 8) & 0xf; + const u32 rm = instruction & 0xf; + if (condition == 0xf || rd_hi == 15 || rd_lo == 15 || rs == 15 || rm == 15) + return false; + + if (condition == 0xe) + return EmitARMMultiplyLong(instruction); + + EmitConditionResult(condition); + TEST(32, R(EAX), R(EAX)); + const FixupBranch predicate_failed = J_CC(CC_Z, Jump::Near); + const bool emitted = EmitARMMultiplyLong(instruction); + ASSERT(emitted); + SetJumpTarget(predicate_failed); + return true; + } + if ((instruction & 0x0e000000) == 0x0a000000 && (instruction >> 28) != 0xf) { const bool link = (instruction & (1U << 24)) != 0; @@ -686,7 +1041,7 @@ bool ARMJitX64::EmitDirectARM(u32 instruction, u32 address, bool* terminal, bool EmitConditionResult(condition); TEST(32, R(EAX), R(EAX)); const FixupBranch predicate_failed = J_CC(CC_Z, Jump::Near); - const bool emitted = EmitARMMemory(instruction, address); + const bool emitted = EmitARMMemory(instruction, address, terminal); ASSERT(emitted); SetJumpTarget(predicate_failed); return true; @@ -707,7 +1062,7 @@ bool ARMJitX64::EmitDirectARM(u32 instruction, u32 address, bool* terminal, bool } if ((instruction & 0x0c000000) == 0x04000000) - return EmitARMMemory(instruction, address); + return EmitARMMemory(instruction, address, terminal); if ((instruction & 0x0e000090) == 0x00000090) return EmitARMHalfwordMemory(instruction, address); @@ -722,6 +1077,73 @@ bool ARMJitX64::EmitDirectARM(u32 instruction, u32 address, bool* terminal, bool return true; } +bool ARMJitX64::EmitARMMultiplyLong(u32 instruction) +{ + if ((instruction & 0x0f8000f0) != 0x00800090) + return false; + + const bool signed_multiply = (instruction & (1U << 22)) != 0; + const bool accumulate = (instruction & (1U << 21)) != 0; + const bool set_flags = (instruction & (1U << 20)) != 0; + const u32 rd_hi = (instruction >> 16) & 0xf; + const u32 rd_lo = (instruction >> 12) & 0xf; + const u32 rs = (instruction >> 8) & 0xf; + const u32 rm = instruction & 0xf; + if (rd_hi == 15 || rd_lo == 15 || rs == 15 || rm == 15) + return false; + + // Read every guest operand before writing either destination so architecturally tolerated + // source/destination aliases behave exactly like ARMCore::ExecuteMultiplyLong. + FlushRegisterCache(); + if (signed_multiply) + { + MOVSX(64, 32, RAX, MStoredRegister(rm)); + MOVSX(64, 32, RCX, MStoredRegister(rs)); + } + else + { + MOV(32, R(EAX), MStoredRegister(rm)); + MOV(32, R(ECX), MStoredRegister(rs)); + } + IMUL(64, RAX, R(RCX)); + + if (accumulate) + { + MOV(32, R(R8), MStoredRegister(rd_hi)); + SHL(64, R(R8), Imm8(32)); + MOV(32, R(R9), MStoredRegister(rd_lo)); + OR(64, R(R8), R(R9)); + ADD(64, R(RAX), R(R8)); + } + + if (set_flags) + { + TEST(64, R(RAX), R(RAX)); + SETcc(CC_S, R(R8)); + SETcc(CC_Z, R(R9)); + MOVZX(32, 8, R8, R(R8)); + MOVZX(32, 8, R9, R(R9)); + SHL(32, R(R8), Imm8(31)); + SHL(32, R(R9), Imm8(30)); + } + + MOV(32, MStoredRegister(rd_lo), R(EAX)); + MOV(64, R(RDX), R(RAX)); + SHR(64, R(RDX), Imm8(32)); + MOV(32, MStoredRegister(rd_hi), R(EDX)); + + if (set_flags) + { + MOV(32, R(ECX), MCPSR()); + AND(32, R(ECX), Imm32(~(ARMCore::CPSR_N | ARMCore::CPSR_Z))); + OR(32, R(ECX), R(R8)); + OR(32, R(ECX), R(R9)); + MOV(32, MCPSR(), R(ECX)); + } + LoadRegisterCache(); + return true; +} + bool ARMJitX64::CanEmitARMDataProcessing(u32 instruction) const { // The special encodings in the data-processing space stay on the exact interpreter path. The @@ -759,24 +1181,18 @@ bool ARMJitX64::CanEmitARMDataProcessing(u32 instruction) const const bool shifted_register_operand = !immediate && (instruction & 0xff0) != 0; const bool shift_by_register = shifted_register_operand && (instruction & (1U << 4)) != 0; const u32 rs = (instruction >> 8) & 0xf; - // Carry-out from a shifted operand is only architecturally visible for logical flag-setting - // operations. Keep those on the interpreter for now; all non-flag-setting ALU forms can use the - // value-only native shifter exactly. - if (shifted_register_operand && (set_flags || !writes_result)) + // Register-controlled shifts have several ARM-only carry corner cases for counts >= 32. Keep + // only those flag-setting forms on the interpreter; immediate shifts can materialize their exact + // shifter carry cheaply below. + if (shift_by_register && (set_flags || !writes_result)) return false; if (shift_by_register && rs == 15) return false; - const bool logical = opcode == 0 || opcode == 1 || opcode == 8 || opcode == 9 || opcode == 0xc || - opcode == 0xd || opcode == 0xe || opcode == 0xf; - const u32 rotate = ((instruction >> 8) & 0xf) * 2; - if (logical && (set_flags || !writes_result) && immediate && rotate != 0) - return false; - return true; } -bool ARMJitX64::EmitARMMemory(u32 instruction, u32 address) +bool ARMJitX64::EmitARMMemory(u32 instruction, u32 address, bool* terminal) { if (!CanEmitARMMemory(instruction)) return false; @@ -878,7 +1294,21 @@ bool ARMJitX64::EmitARMMemory(u32 instruction, u32 address) SHL(32, R(ECX), Imm8(3)); ROR(32, R(EAX), R(ECX)); } - MOV(32, MRegister(rd), R(EAX)); + if (rd == 15) + { + // ARMv5 LDR PC is an interworking branch. Preserve bit zero as the new Thumb state and + // branch to the aligned target. This is the exact operation used by Starlet's high IRQ and + // reset vectors, and keeping it in native code avoids a generic decoder call per interrupt. + MOV(32, R(ABI_PARAM3), R(EAX)); + MOV(32, R(ABI_PARAM2), Imm32(address)); + MOV(64, R(ABI_PARAM1), ImmPtr(this)); + ABI_CallFunction(LoadPC); + *terminal = true; + } + else + { + MOV(32, MRegister(rd), R(EAX)); + } } else { @@ -907,7 +1337,10 @@ bool ARMJitX64::CanEmitARMMemory(u32 instruction) const const u32 rd = (instruction >> 12) & 0xf; const bool register_offset = (instruction & (1U << 25)) != 0; const u32 rm = instruction & 0xf; - return rd != 15 && !(rn == 15 && (!preindex || writeback)) && !(load && writeback && rn == rd) && + const bool direct_pc_load = load && rd == 15 && (instruction >> 28) == 0xe && preindex && + !writeback && (instruction & (1U << 22)) == 0; + return (rd != 15 || direct_pc_load) && !(rn == 15 && (!preindex || writeback)) && + !(load && writeback && rn == rd) && !(register_offset && (instruction & (1U << 4)) != 0) && !(register_offset && rm == 15); } @@ -1123,11 +1556,21 @@ void ARMJitX64::EmitARMDataProcessing(u32 instruction) const u32 rd = (instruction >> 12) & 0xf; const u32 rm = instruction & 0xf; const bool writes_result = opcode < 8 || opcode > 0xb; + const bool logical = opcode == 0 || opcode == 1 || opcode == 8 || opcode == 9 || opcode == 0xc || + opcode == 0xd || opcode == 0xe || opcode == 0xf; + const bool logical_flags = logical && (set_flags || !writes_result); + bool logical_carry_known = false; if (immediate) { const u32 rotate = ((instruction >> 8) & 0xf) * 2; - MOV(32, R(EDX), Imm32(std::rotr(instruction & 0xff, rotate))); + const u32 operand = std::rotr(instruction & 0xff, rotate); + MOV(32, R(EDX), Imm32(operand)); + if (logical_flags && rotate != 0) + { + MOV(32, R(R10), Imm32(operand >> 31)); + logical_carry_known = true; + } } else { @@ -1168,6 +1611,21 @@ void ARMJitX64::EmitARMDataProcessing(u32 instruction) else { const u32 amount = (instruction >> 7) & 0x1f; + if (logical_flags) + { + // ARM's logical S forms take C from the barrel shifter. Save that source bit before EDX + // is shifted; rotate zero preserves the old C only for the unshifted LSL #0 encoding, + // which never enters this branch. + MOV(32, R(R10), R(EDX)); + if (shift_type == 0) + SHR(32, R(R10), Imm8(32 - amount)); + else if (shift_type == 1 || shift_type == 2) + SHR(32, R(R10), Imm8(amount == 0 ? 31 : amount - 1)); + else if (amount != 0) + SHR(32, R(R10), Imm8(amount - 1)); + AND(32, R(R10), Imm8(1)); + logical_carry_known = true; + } if (shift_type == 0) { if (amount != 0) @@ -1253,6 +1711,8 @@ void ARMJitX64::EmitARMDataProcessing(u32 instruction) { if (arithmetic) EmitArithmeticFlags(opcode == 2 || opcode == 3 || opcode == 0xa); + else if (logical_carry_known) + EmitLogicalFlagsWithCarry(EAX, R10); else EmitLogicalFlags(EAX); } @@ -1869,6 +2329,8 @@ void ARMJitX64::EmitFastmemAddress(std::vector* slow_paths, u32 acc { CMP(32, R(EDX), Imm32(0x00)); const FixupBranch profiled_read_page_00 = J_CC(CC_E, Jump::Near); + CMP(32, R(EDX), Imm32(0x10)); + const FixupBranch irq_vector_read_page_10 = J_CC(CC_E, Jump::Near); CMP(32, R(EDX), Imm32(0x12)); const FixupBranch profiled_read_page_12 = J_CC(CC_E, Jump::Near); CMP(32, R(EDX), Imm32(0x14)); @@ -1878,6 +2340,7 @@ void ARMJitX64::EmitFastmemAddress(std::vector* slow_paths, u32 acc CMP(32, R(EDX), Imm32(0x1e)); slow_paths->push_back(J_CC(CC_NE, Jump::Near)); SetJumpTarget(profiled_read_page_00); + SetJumpTarget(irq_vector_read_page_10); SetJumpTarget(profiled_read_page_12); SetJumpTarget(profiled_read_page_14); SetJumpTarget(profiled_read_page_19); @@ -2368,6 +2831,11 @@ OpArg ARMJitX64::MExecutedInstructions() const return MDisp(R15, m_executed_instructions_offset); } +OpArg ARMJitX64::MDomainAccessControl() const +{ + return MDisp(R15, m_domain_access_control_offset); +} + void ARMJitX64::FallbackThumb(ARMJitX64* jit, u16 instruction, u32 address) { ARMCore* const core = &jit->m_core; @@ -2417,6 +2885,23 @@ void ARMJitX64::WritePSR(ARMJitX64* jit, u32 spsr, u32 field_mask, u32 value) jit->m_core.WritePSR(spsr != 0, field_mask, value); } +void ARMJitX64::InvalidateTLB(ARMJitX64* jit, u32 address) +{ + ARMCore& core = jit->m_core; + core.m_instruction_address = address; + core.m_pc_written = false; + core.InvalidateTLB(); + core.m_registers[15] = address + 4; +} + +void ARMJitX64::LoadPC(ARMJitX64* jit, u32 address, u32 target) +{ + ARMCore& core = jit->m_core; + core.m_instruction_address = address; + core.m_pc_written = false; + core.WritePC(target, true); +} + void ARMJitX64::ExecuteUserBankBlockTransfer(ARMJitX64* jit, u32 instruction, u32 address) { ARMCore& core = jit->m_core; @@ -2441,9 +2926,11 @@ void ARMJitX64::ExceptionReturn(ARMJitX64* jit, u32 target) } u32 ARMJitX64::ReadMemorySlow(ARMJitX64* jit, u32 physical_address, u32 access_size, - u32 byte_offset) + u32 byte_offset) { ++jit->m_slow_read_count; + if ((jit->m_slow_read_count & 0xff) == 0) + ++jit->m_slow_memory_address_samples[physical_address & ~3ULL]; switch (ClassifySlowMemoryAddress(physical_address)) { case SlowMemoryRegion::RAM: @@ -2475,6 +2962,8 @@ u32 ARMJitX64::ReadMemorySlow(ARMJitX64* jit, u32 physical_address, u32 access_s void ARMJitX64::WriteMemorySlow(ARMJitX64* jit, u32 physical_address, u32 access_size, u32 value) { ++jit->m_slow_write_count; + if ((jit->m_slow_write_count & 0xff) == 0) + ++jit->m_slow_memory_address_samples[(1ULL << 32) | (physical_address & ~3ULL)]; switch (ClassifySlowMemoryAddress(physical_address)) { case SlowMemoryRegion::RAM: @@ -2505,6 +2994,18 @@ void ARMJitX64::WriteMemorySlow(ARMJitX64* jit, u32 physical_address, u32 access core.m_bus.Write32(physical_address & ~3U, value); } +std::vector> ARMJitX64::TakeHotSlowMemorySamples(size_t maximum_count) +{ + std::vector> sorted(m_slow_memory_address_samples.begin(), + m_slow_memory_address_samples.end()); + m_slow_memory_address_samples.clear(); + std::ranges::sort(sorted, {}, [](const auto& entry) { return entry.second; }); + if (sorted.size() > maximum_count) + sorted.erase(sorted.begin(), sorted.end() - maximum_count); + std::ranges::reverse(sorted); + return sorted; +} + u32 ARMJitX64::TranslateAddress(ARMJitX64* jit, u32 address) { ++jit->m_address_translation_count; diff --git a/Source/Core/Core/IOS/Starlet/ARMJitX64.h b/Source/Core/Core/IOS/Starlet/ARMJitX64.h index 5d4de4064d..86bd9e64c4 100644 --- a/Source/Core/Core/IOS/Starlet/ARMJitX64.h +++ b/Source/Core/Core/IOS/Starlet/ARMJitX64.h @@ -35,6 +35,53 @@ public: u64 GetExecutedInstructions() const { return m_executed_instructions; } u64 GetNativeExecutedInstructions() const { return m_native_executed_instructions; } size_t GetCompiledBlockCount() const { return m_blocks.size(); } + u64 GetLifetimeCompiledBlockCount() const { return m_lifetime_compiled_block_count; } + u64 GetBlockLookupCount() const { return m_block_lookup_count; } + u64 GetBlockCacheHitCount() const { return m_block_cache_hit_count; } + u64 GetDispatchSlowCount() const { return m_dispatch_slow_count; } + u64 GetDispatchStaleGenerationCount() const { return m_dispatch_stale_generation_count; } + u64 GetDispatchPhysicalAliasCount() const { return m_dispatch_physical_alias_count; } + u64 GetDispatchCollisionCount() const { return m_dispatch_collision_count; } + u64 GetDispatchKeyMissCount() const { return m_dispatch_key_miss_count; } + u64 GetDispatchPageMissCount() const { return m_dispatch_page_miss_count; } + u64 GetDispatchEmptyEntryCount() const { return m_dispatch_empty_entry_count; } + u64 GetDispatchGeneratedKeyMismatchCount() const + { + return m_dispatch_generated_key_mismatch_count; + } + u64 GetDispatchGeneratedSetMismatchCount() const + { + return m_dispatch_generated_set_mismatch_count; + } + u64 GetDispatchKeyMissWithExactMatchCount() const + { + return m_dispatch_key_miss_with_exact_match_count; + } + u64 GetDispatchKeyMissWithPhysicalAliasCount() const + { + return m_dispatch_key_miss_with_physical_alias_count; + } + u64 GetDispatchKeyMissWithSamePhysicalPageCount() const + { + return m_dispatch_key_miss_with_same_physical_page_count; + } + u64 GetDispatchKeyMissWithEmptySlotCount() const + { + return m_dispatch_key_miss_with_empty_slot_count; + } + u64 GetDispatchKeyMissWithFullSetCount() const + { + return m_dispatch_key_miss_with_full_set_count; + } + u64 GetDispatchKeyMissWithStaleTranslationCount() const + { + return m_dispatch_key_miss_with_stale_translation_count; + } + u64 GetFastEntryEmptyInsertCount() const { return m_fast_entry_empty_insert_count; } + u64 GetFastEntryReplacementCount() const { return m_fast_entry_replacement_count; } + u64 GetInstructionCacheClearCount() const { return m_instruction_cache_clear_count; } + u64 GetCodeSpaceClearCount() const { return m_code_space_clear_count; } + u64 GetDiscardedBlockCount() const { return m_discarded_block_count; } u64 GetAddressTranslationCount() const { return m_address_translation_count; } u64 GetSlowReadCount() const { return m_slow_read_count; } u64 GetSlowWriteCount() const { return m_slow_write_count; } @@ -62,6 +109,7 @@ public: m_slow_sram_page_write_access_count[page] : 0; } + std::vector> TakeHotSlowMemorySamples(size_t maximum_count); private: using RunEntry = u64 (*)(u32); @@ -84,21 +132,54 @@ private: struct FastEntry { u32 key = 0xffffffff; - u32 tlb_generation = 0; + u32 physical_page = 0xffffffff; const u8* entry = nullptr; }; + static_assert(sizeof(FastEntry) == 16); + + // TLB maintenance invalidates translations, not every decoded instruction on a page. Keep one + // generation-tagged physical identity per 1 KiB ARM926 TLB granule so the first block after a + // flush performs the page-table walk and the remaining blocks can safely retain native code. + struct alignas(64) FastTranslationEntry + { + u32 virtual_page = 0xffffffff; + u32 physical_page = 0xffffffff; + u32 tlb_generation = 0; + u32 translation_table_base = 0xffffffff; + const u8* first_descriptor = nullptr; + const u8* second_descriptor = nullptr; + u32 first_descriptor_value = 0; + u32 second_descriptor_value = 0; + std::array padding{}; + }; + static_assert(sizeof(FastTranslationEntry) == 64); + + enum class DispatchReason : u32 + { + Translation, + KeyMiss, + PageMiss, + EmptyEntry, + }; void PoisonMemory() override; + void RequestClear(); + void ClearCodeCache(); + void ClearForCodeSpace(); void GenerateDispatcher(); static u32 MakeFastEntryKey(u32 address, u32 control, u32 process_id); - static size_t GetFastEntryIndex(u32 key); - static const u8* Dispatch(ARMJitX64* jit); - Block* GetOrCompileBlock(u32 address); + static size_t GetFastEntrySetIndex(u32 key); + static const u8* Dispatch(ARMJitX64* jit, DispatchReason reason, u32 generated_key, + u32 generated_set_offset); + Block* GetOrCompileBlock(u32 address, u32* physical_address_out, + const u8** first_descriptor_out, u32* first_descriptor_value_out, + const u8** second_descriptor_out, u32* second_descriptor_value_out); Block CompileBlock(u32 address, bool thumb); bool EmitDirectARM(u32 instruction, u32 address, bool* terminal, bool* dispatcher_exit); + bool EmitARMMultiplyLong(u32 instruction); bool CanEmitARMDataProcessing(u32 instruction) const; bool CanEmitARMMemory(u32 instruction) const; - bool EmitARMMemory(u32 instruction, u32 address); + bool EmitARMMemory(u32 instruction, u32 address, bool* terminal); bool EmitARMHalfwordMemory(u32 instruction, u32 address); bool EmitARMBlockTransfer(u32 instruction, u32 address, bool* terminal); void EmitARMDataProcessing(u32 instruction); @@ -137,11 +218,14 @@ private: Gen::OpArg MWaitingForMemoryPoll() const; Gen::OpArg MYieldRequested() const; Gen::OpArg MExecutedInstructions() const; + Gen::OpArg MDomainAccessControl() const; static void FallbackThumb(ARMJitX64* jit, u16 instruction, u32 address); static void FallbackARM(ARMJitX64* jit, u32 instruction, u32 address); static u32 ReadSPSR(ARMJitX64* jit); static void WritePSR(ARMJitX64* jit, u32 spsr, u32 field_mask, u32 value); + static void InvalidateTLB(ARMJitX64* jit, u32 address); + static void LoadPC(ARMJitX64* jit, u32 address, u32 target); static void ExecuteUserBankBlockTransfer(ARMJitX64* jit, u32 instruction, u32 address); static void ExecuteThumbPushPop(ARMJitX64* jit, u16 instruction, u32 address); static void ExceptionReturn(ARMJitX64* jit, u32 target); @@ -154,11 +238,17 @@ private: static constexpr size_t CODE_SIZE = 32 * 1024 * 1024; static constexpr u32 MAX_BLOCK_INSTRUCTIONS = 32; - static constexpr size_t FAST_ENTRY_COUNT = 1 << 16; + static constexpr size_t FAST_ENTRY_SET_COUNT = 1 << 16; + static constexpr size_t FAST_ENTRY_WAYS = 4; + static constexpr size_t FAST_ENTRY_COUNT = FAST_ENTRY_SET_COUNT * FAST_ENTRY_WAYS; + static_assert((FAST_ENTRY_WAYS & (FAST_ENTRY_WAYS - 1)) == 0); + static constexpr size_t FAST_TRANSLATION_ENTRY_COUNT = 1 << 12; ARMCore& m_core; std::unordered_map m_blocks; - std::array m_fast_entries{}; + alignas(64) std::array m_fast_entries{}; + std::array m_fast_entry_next_victim{}; + std::array m_fast_translations{}; u8* m_fastmem_base = nullptr; u8* m_sram_base = nullptr; const bool* m_boot0_mapped = nullptr; @@ -169,6 +259,29 @@ private: u8* m_block_code_begin = nullptr; u64 m_executed_instructions = 0; u64 m_native_executed_instructions = 0; + u64 m_lifetime_compiled_block_count = 0; + u64 m_block_lookup_count = 0; + u64 m_block_cache_hit_count = 0; + u64 m_dispatch_slow_count = 0; + u64 m_dispatch_stale_generation_count = 0; + u64 m_dispatch_physical_alias_count = 0; + u64 m_dispatch_collision_count = 0; + u64 m_dispatch_key_miss_count = 0; + u64 m_dispatch_page_miss_count = 0; + u64 m_dispatch_empty_entry_count = 0; + u64 m_dispatch_generated_key_mismatch_count = 0; + u64 m_dispatch_generated_set_mismatch_count = 0; + u64 m_dispatch_key_miss_with_exact_match_count = 0; + u64 m_dispatch_key_miss_with_physical_alias_count = 0; + u64 m_dispatch_key_miss_with_same_physical_page_count = 0; + u64 m_dispatch_key_miss_with_empty_slot_count = 0; + u64 m_dispatch_key_miss_with_full_set_count = 0; + u64 m_dispatch_key_miss_with_stale_translation_count = 0; + u64 m_fast_entry_empty_insert_count = 0; + u64 m_fast_entry_replacement_count = 0; + u64 m_instruction_cache_clear_count = 0; + u64 m_code_space_clear_count = 0; + u64 m_discarded_block_count = 0; u64 m_address_translation_count = 0; u64 m_slow_read_count = 0; u64 m_slow_write_count = 0; @@ -184,6 +297,7 @@ private: u64 m_slow_sram_high_access_count = 0; std::array m_slow_sram_page_read_access_count{}; std::array m_slow_sram_page_write_access_count{}; + std::unordered_map m_slow_memory_address_samples; bool m_is_running = false; bool m_clear_pending = false; s32 m_registers_offset = 0; @@ -195,6 +309,8 @@ private: s32 m_yield_requested_offset = 0; s32 m_executed_instructions_offset = 0; s32 m_control_offset = 0; + s32 m_translation_table_base_offset = 0; + s32 m_domain_access_control_offset = 0; s32 m_process_id_offset = 0; s32 m_tlb_generation_offset = 0; u32 m_compile_instruction_count = 0; diff --git a/Source/Core/Core/IOS/Starlet/Starlet.cpp b/Source/Core/Core/IOS/Starlet/Starlet.cpp index 5ba431268b..b13564c3e5 100644 --- a/Source/Core/Core/IOS/Starlet/Starlet.cpp +++ b/Source/Core/Core/IOS/Starlet/Starlet.cpp @@ -296,6 +296,32 @@ void Starlet::RunSlice(s64 cycles_late) m_core->GetCPSR(), m_system.GetWiiIPC().ReadStarletRegister(0x38), m_system.GetWiiIPC().ReadStarletRegister(0x3c), m_system.GetWiiIPC().ReadStarletRegister(0x40)); + INFO_LOG_FMT(IOS, + "Starlet JIT cache: live-blocks={} lifetime-compiles={} lookups={} hits={} " + "slow-dispatches={} stale-generations={} physical-aliases={} collisions={} " + "key-misses={} page-misses={} empty-entries={} " + "generated-key-mismatches={} generated-set-mismatches={} exact-key-misses={} " + "key-aliases={} key-same-pages={} key-empty-slots={} key-full-sets={} " + "key-stale-translations={} empty-inserts={} replacements={} " + "icache-clears={} code-space-clears={} discarded-blocks={}", + m_core->GetJitCompiledBlockCount(), m_core->GetJitLifetimeCompiledBlockCount(), + m_core->GetJitBlockLookupCount(), m_core->GetJitBlockCacheHitCount(), + m_core->GetJitDispatchSlowCount(), m_core->GetJitDispatchStaleGenerationCount(), + m_core->GetJitDispatchPhysicalAliasCount(), m_core->GetJitDispatchCollisionCount(), + m_core->GetJitDispatchKeyMissCount(), m_core->GetJitDispatchPageMissCount(), + m_core->GetJitDispatchEmptyEntryCount(), + m_core->GetJitDispatchGeneratedKeyMismatchCount(), + m_core->GetJitDispatchGeneratedSetMismatchCount(), + m_core->GetJitDispatchKeyMissWithExactMatchCount(), + m_core->GetJitDispatchKeyMissWithPhysicalAliasCount(), + m_core->GetJitDispatchKeyMissWithSamePhysicalPageCount(), + m_core->GetJitDispatchKeyMissWithEmptySlotCount(), + m_core->GetJitDispatchKeyMissWithFullSetCount(), + m_core->GetJitDispatchKeyMissWithStaleTranslationCount(), + m_core->GetJitFastEntryEmptyInsertCount(), + m_core->GetJitFastEntryReplacementCount(), + m_core->GetJitInstructionCacheClearCount(), m_core->GetJitCodeSpaceClearCount(), + m_core->GetJitDiscardedBlockCount()); std::array hot_sram_read_pages{}; std::array hot_sram_read_page_counts{}; std::array hot_sram_write_pages{}; @@ -335,6 +361,12 @@ void Starlet::RunSlice(s64 cycles_late) hot_sram_write_page_counts[1], hot_sram_write_pages[2], hot_sram_write_page_counts[2], hot_sram_write_pages[3], hot_sram_write_page_counts[3]); + for (const auto& [key, samples] : m_core->TakeHotJitSlowMemorySamples(8)) + { + const bool write = (key >> 32) != 0; + INFO_LOG_FMT(IOS, "Starlet slow memory {} address={:#010x} samples={}", + write ? "write" : "read", static_cast(key), samples); + } const ARMCore::HotPCSample current = m_core->GetCurrentPCSample(); INFO_LOG_FMT(IOS, "Starlet current PC {:#010x} {} instruction={:#010x}", current.address, current.thumb ? "Thumb" : "ARM", current.instruction); diff --git a/Source/Core/Core/IOS/Starlet/StarletMemory.cpp b/Source/Core/Core/IOS/Starlet/StarletMemory.cpp index 4044b2cb93..e7c01e559c 100644 --- a/Source/Core/Core/IOS/Starlet/StarletMemory.cpp +++ b/Source/Core/Core/IOS/Starlet/StarletMemory.cpp @@ -189,13 +189,6 @@ constexpr u32 OHCI_PORT_CHANGE_MASK = 0x001f0000; constexpr u32 OHCI_FRAME_CYCLES = 243000; constexpr u16 OHCI1_ATTACH_DELAY_FRAMES = 100; constexpr u64 WIIMOTE_UPDATE_CYCLES = 243000000 / Wiimote::UPDATE_FREQ; -// Poll host controls at Dolphin's normal 200 Hz, but do not wake the -// interpreted IOS Bluetooth stack for every poll. A 30 Hz steady HID stream -// leaves substantially more host time for Broadway; button transitions bypass -// this throttle below so presses and releases still reach IOS promptly. -constexpr u32 WIIMOTE_REPORT_FREQUENCY = 30; -constexpr u64 WIIMOTE_REPORT_CYCLES = 243000000 / WIIMOTE_REPORT_FREQUENCY; -static_assert(WIIMOTE_REPORT_FREQUENCY <= Wiimote::UPDATE_FREQ); // Hollywood completes the internal OHCI1 port reset before IOS's first 2 ms // poll. Using the generic 10 ms upper-bound timing leaves IOS with RHSC masked // when PRSC arrives. @@ -334,6 +327,7 @@ constexpr u32 HW_USBFRCRST = HW_BASE + 0x88; constexpr u32 HW_SRNPROT = HW_BASE + 0x60; constexpr u32 HW_AHBPROT = HW_BASE + 0x64; constexpr u32 HW_TIMER = HW_BASE + 0x10; +constexpr u64 TIMER_CLOCK_DIVISOR = 128; constexpr u32 HW_ALARM = HW_BASE + 0x14; constexpr u32 HW_GPIO_ENABLE = HW_BASE + 0xdc; constexpr u32 HW_GPIO_OUT = HW_BASE + 0xe0; @@ -683,6 +677,7 @@ void StarletMemory::Reset() for (size_t controller = 0; controller < OHCI_BASES.size(); ++controller) ResetOHCIController(controller); m_arm_cycles = 0; + m_timer_offset = 0; m_boot0_mapped = true; m_sram_split_mode = false; // Retail Hollywood production revision. Early firmware branches on this @@ -786,6 +781,7 @@ void StarletMemory::DoState(PointerWrap& p) p.Do(m_seeprom_write_pending); p.Do(m_seeprom_write_all); p.Do(m_arm_cycles); + p.Do(m_timer_offset); p.Do(m_initialized); p.Do(m_boot0_mapped); p.Do(m_sram_split_mode); @@ -799,10 +795,11 @@ void StarletMemory::DoState(PointerWrap& p) u32 StarletMemory::GetTimer() const { - // Hollywood's 19.2 MHz timer is clocked at 32/405 of the 243 MHz Starlet - // clock. - const u64 timer = (m_arm_cycles / 405) * 32 + ((m_arm_cycles % 405) * 32) / 405; - return static_cast(timer); + // Hollywood's timer runs at Starlet / 128 (1.8984375 MHz at 243 MHz). + // Keep the divider phase across scheduler slices. The writable counter has + // its own offset so IOS resetting HW_TIMER cannot rewind peripheral time. + // https://wiibrew.org/wiki/Hardware/Starlet_Timer + return static_cast(m_arm_cycles / TIMER_CLOCK_DIVISOR) + m_timer_offset; } std::optional StarletMemory::TryReadBroadwayResetInstruction(u32 address) const @@ -895,6 +892,32 @@ const bool* StarletMemory::GetFastmemSRAMSplitMode() const return &m_sram_split_mode; } +const u8* StarletMemory::GetDirectMemoryPointer(u32 address, u32 size) const +{ + if (size == 0 || address > std::numeric_limits::max() - (size - 1)) + return nullptr; + + const u32 last_address = address + size - 1; + if (IsMemoryAddress(address) && IsMemoryAddress(last_address)) + { + if (u8* const base = m_system.GetMemory().GetPhysicalBase()) + return base + address; + } + + // Page tables can live in Hollywood SRAM. Respect the same boot0 overlay and A/B split mapping + // as normal bus reads, and only expose a pointer when the whole descriptor is contiguous. + if (IsSRAMWindowAddress(address) && IsSRAMWindowAddress(last_address) && + !IsBootROMAddress(address) && !IsBootROMAddress(last_address)) + { + const u32 offset = GetSRAMOffset(address); + const u32 end_offset = GetSRAMOffset(last_address); + if (offset != INVALID_SRAM_OFFSET && end_offset == offset + size - 1) + return m_sram.data() + offset; + } + + return nullptr; +} + bool StarletMemory::IsMemoryAddress(u32 address) { return address < Memory::MEM1_SIZE_RETAIL || @@ -1677,24 +1700,17 @@ void StarletMemory::UpdateWiimotes() next_calls[i] = m_wiimotes[i]->PrepareInput(&states[i]); } - const u64 previous_update_cycles = - m_arm_cycles >= WIIMOTE_UPDATE_CYCLES ? m_arm_cycles - WIIMOTE_UPDATE_CYCLES : 0; - const bool report_due = - m_arm_cycles / WIIMOTE_REPORT_CYCLES != previous_update_cycles / WIIMOTE_REPORT_CYCLES; - + // Deliver input at the normal Wii Remote cadence, even when buttons have not + // changed. KPAD repeat timing depends on fresh reports; starving KPADRead can + // also leave a previous trigger visible across successive menu frames. for (size_t i = 0; i < m_wiimotes.size(); ++i) { if (!m_wiimotes[i]) continue; - const bool button_changed = states[i].buttons.hex != m_last_wiimote_buttons[i]; - if (next_calls[i] != IOS::HLE::WiimoteDevice::NextUpdateInputCall::Update || report_due || - button_changed) - { - m_wiimotes[i]->UpdateInput(next_calls[i], states[i]); - if (next_calls[i] == IOS::HLE::WiimoteDevice::NextUpdateInputCall::Update) - m_last_wiimote_buttons[i] = states[i].buttons.hex; - } + m_wiimotes[i]->UpdateInput(next_calls[i], states[i]); + if (next_calls[i] == IOS::HLE::WiimoteDevice::NextUpdateInputCall::Update) + m_last_wiimote_buttons[i] = states[i].buttons.hex; } } @@ -3326,6 +3342,22 @@ u16 StarletMemory::Read16(u32 address) u32 StarletMemory::Read32(u32 address) { + // ARM IOS accesses the Hollywood timer and interrupt controller almost exclusively as aligned + // words. Falling back to ARMBus::Read32 decomposes each access into four virtual Read8 calls; + // every byte then repeats the complete device-range decoder and IPC register switch. Preserve + // exactly the same live values while resolving these two hottest register families once. + if ((address & 3) == 0) + { + if (address == HW_TIMER) + return GetTimer(); + if ((address >= HW_BASE && address <= HW_BASE + 0x0c) || + (address >= HW_BASE + 0x30 && address <= HW_BASE + 0x40) || address == HW_AHBPROT || + address == HW_RESETS) + { + return m_system.GetWiiIPC().ReadStarletRegister(address - HW_BASE); + } + } + if ((address & 3) == 0 && IsDIAddress(address)) { MMIO::Mapping* const mmio = m_system.GetMemory().GetMMIOMapping(); @@ -3513,7 +3545,7 @@ void StarletMemory::Write8(u32 address, u8 value) else if (word_address == HW_OTPCMD) HandleOTPCommand(word); else if (word_address == HW_TIMER) - m_arm_cycles = static_cast(word) * 405 / 32; + m_timer_offset = word - static_cast(m_arm_cycles / TIMER_CLOCK_DIVISOR); else if (word_address == HW_ALARM) { // HW_ALARM is a comparator, not an interrupt acknowledgement register. @@ -3602,6 +3634,29 @@ void StarletMemory::Write16(u32 address, u16 value) void StarletMemory::Write32(u32 address, u32 value) { + // Match the aligned read fast path above. WriteRegister keeps the byte-addressable backing image + // coherent, while the exact device handlers retain W1C interrupt flags, reset side effects and + // the Starlet/Broadway scheduling boundary of the four-byte Write8 path. + if ((address & 3) == 0 && address == HW_TIMER) + { + WriteRegister(address, value); + m_timer_offset = value - static_cast(m_arm_cycles / TIMER_CLOCK_DIVISOR); + return; + } + if ((address & 3) == 0 && ((address >= HW_BASE && address <= HW_BASE + 0x0c) || + (address >= HW_BASE + 0x30 && address <= HW_BASE + 0x40) || + address == HW_AHBPROT || address == HW_RESETS)) + { + WriteRegister(address, value); + m_system.GetWiiIPC().WriteStarletRegister(address - HW_BASE, value); + if (address == HW_BASE + 0x0c && (value & 0x09) != 0) + { + if (Starlet* const starlet = m_system.GetStarlet()) + starlet->YieldForIPC(); + } + return; + } + if ((address & 3) == 0 && IsDIAddress(address)) { MMIO::Mapping* const mmio = m_system.GetMemory().GetMMIOMapping(); diff --git a/Source/Core/Core/IOS/Starlet/StarletMemory.h b/Source/Core/Core/IOS/Starlet/StarletMemory.h index 64816a32db..ec3c4c5268 100644 --- a/Source/Core/Core/IOS/Starlet/StarletMemory.h +++ b/Source/Core/Core/IOS/Starlet/StarletMemory.h @@ -69,6 +69,7 @@ public: u8* GetFastmemSRAMBase() const override; const bool* GetFastmemBoot0Mapped() const override; const bool* GetFastmemSRAMSplitMode() const override; + const u8* GetDirectMemoryPointer(u32 address, u32 size) const override; u64 GetCycles() const { return m_arm_cycles; } std::optional TryReadBroadwayResetInstruction(u32 address) const; @@ -302,6 +303,7 @@ private: bool m_seeprom_write_pending = false; bool m_seeprom_write_all = false; u64 m_arm_cycles = 0; + u32 m_timer_offset = 0; bool m_initialized = false; bool m_boot0_mapped = true; bool m_sram_split_mode = false; diff --git a/Source/Core/Core/PowerPC/Jit64/Jit.cpp b/Source/Core/Core/PowerPC/Jit64/Jit.cpp index 1248b1ab01..e3b0c043cd 100644 --- a/Source/Core/Core/PowerPC/Jit64/Jit.cpp +++ b/Source/Core/Core/PowerPC/Jit64/Jit.cpp @@ -3,7 +3,11 @@ #include "Core/PowerPC/Jit64/Jit.h" +#include +#include +#include #include +#include #include #include #include @@ -18,6 +22,7 @@ #endif #include "Common/CommonTypes.h" +#include "Common/FileUtil.h" #include "Common/GekkoDisassembler.h" #include "Common/HostDisassembler.h" #include "Common/IOFile.h" @@ -51,6 +56,111 @@ using namespace Gen; using namespace PowerPC; +namespace +{ +// Temporary, opt-in event diagnostics. Addresses and expected instructions come from a local +// tracepoint file, not from a title-specific emulation rule. No guest state is modified. +struct PPCEventTrace +{ + std::map points; + File::IOFile output; + std::chrono::steady_clock::time_point start; + std::chrono::steady_clock::time_point last_flush; + u32 rows = 0; + + void Init() + { + points.clear(); + rows = 0; + const char* config_path = std::getenv("DOLPHIN_PPC_EVENT_TRACE"); + const char* output_path = std::getenv("DOLPHIN_PPC_EVENT_TRACE_OUTPUT"); + if (!config_path || !output_path) + return; + std::string config; + if (!File::ReadFileToString(config_path, config)) + return; + std::istringstream input(config); + u32 pc, instruction; + while (input >> std::hex >> pc >> instruction) + { + if ((pc & 3) || points.size() >= 32) + { + points.clear(); + return; + } + points.emplace(pc, instruction); + } + if (points.empty() || !output.Open(output_path, "ab", File::SharedAccess::Read)) + { + points.clear(); + return; + } + output.WriteString("seq,wall_us,ticks,pc,lr,ctr,r3,r4,r5,r6,r7,r8," + "mem0,mem1,mem2,mem3,mem4,mem5,mem6\n"); + output.Flush(); + start = last_flush = std::chrono::steady_clock::now(); + } +}; + +PPCEventTrace s_ppc_event_trace; + +// Inspect only the standard cached MEM1/MEM2 aliases. In particular, don't use an MMU load here: +// it could populate a cache line or change PLRU just by observing the packet. +std::optional PeekEventWord(Core::System& system, u32 address) +{ + if ((address & 3) || (address >> 28 != 8 && address >> 28 != 9)) + return std::nullopt; + const u32 physical = address & 0x3fffffff; + auto& memory = system.GetMemory(); + if (!(physical < memory.GetRamSizeReal() && memory.GetRamSizeReal() - physical >= 4) && + !(physical >= 0x10000000 && physical - 0x10000000 < memory.GetExRamSizeReal() && + memory.GetExRamSizeReal() - (physical - 0x10000000) >= 4)) + return std::nullopt; + const auto& ppc = system.GetPowerPC().GetPPCState(); + const UReg_HID0 hid0{.Hex = ppc.spr[SPR_HID0]}; + if (ppc.m_enable_dcache && hid0.DCE) + { + const u32 set = (physical >> 5) & 127; + const auto& cache = ppc.dCache; + for (u32 way = 0; way < 8; ++way) + { + if ((cache.valid[set] & (1 << way)) && cache.addrs[set][way] == (physical & ~31U)) + return Common::swap32(cache.data[set][way][(physical & 31) / 4]); + } + } + const u8* data = memory.GetPointerForRange(physical, 4); + return data ? std::optional(Common::swap32(data)) : std::nullopt; +} + +void RecordPPCEvent(Core::System* system, u32 pc) +{ + auto& trace = s_ppc_event_trace; + if (!trace.output.IsOpen()) + return; + const auto now = std::chrono::steady_clock::now(); + const auto& ppc = system->GetPowerPC().GetPPCState(); + std::string row = + fmt::format("{},{},{},{:08x},{:08x},{:08x}", ++trace.rows, + std::chrono::duration_cast(now - trace.start).count(), + system->GetCoreTiming().GetTicks(), pc, LR(ppc), CTR(ppc)); + for (u32 reg = 3; reg <= 8; ++reg) + row += fmt::format(",{:08x}", ppc.gpr[reg]); + for (u32 word = 0; word < 7; ++word) + { + const auto value = PeekEventWord(*system, ppc.gpr[4] + word * 4); + row += value ? fmt::format(",{:08x}", *value) : ",NA"; + } + row += '\n'; + if (!trace.output.WriteString(row) || trace.rows >= 300000) + trace.output.Close(); + else if (now - trace.last_flush >= std::chrono::milliseconds(250)) + { + trace.output.Flush(); + trace.last_flush = now; + } +} +} // namespace + // Dolphin's PowerPC->x86_64 JIT dynamic recompiler // Written mostly by ector (hrydgard) // Features: @@ -255,6 +365,7 @@ bool Jit64::BackPatch(SContext* ctx) void Jit64::Init() { + s_ppc_event_trace.Init(); InitFastmemArena(); RefreshConfig(); @@ -338,6 +449,9 @@ void Jit64::ResetFreeMemoryRanges() void Jit64::Shutdown() { + if (s_ppc_event_trace.output.IsOpen()) + s_ppc_event_trace.output.Close(); + s_ppc_event_trace.points.clear(); FreeCodeSpace(); auto& memory = m_system.GetMemory(); @@ -824,6 +938,19 @@ void Jit64::Jit(u32 em_address, bool clear_cache_and_retry_on_failure) } } + if (!s_ppc_event_trace.points.empty()) + { + // Force tracepoints to be block entries, where all guest registers are materialized. Avoid + // following branches past these boundaries. Ordinary runs retain their normal optimizations. + analyzer.ClearOption(PPCAnalyst::PPCAnalyzer::OPTION_BRANCH_FOLLOW); + const auto next = s_ppc_event_trace.points.lower_bound(em_address); + if (next != s_ppc_event_trace.points.end()) + { + const size_t distance = (next->first - em_address) / 4; + block_size = std::min(block_size, distance == 0 ? size_t{1} : distance); + } + } + // Analyze the block, collect all instructions it is made of (including inlining, // if that is enabled), reorder instructions for optimal performance, and join joinable // instructions. @@ -926,6 +1053,15 @@ bool Jit64::DoJit(u32 em_address, JitBlock* b, u32 nextPC) // TODO: Test if this or AlignCode16 make a difference from GetCodePtr b->normalEntry = AlignCode4(); + const auto tracepoint = s_ppc_event_trace.points.find(em_address); + if (tracepoint != s_ppc_event_trace.points.end() && + m_code_buffer[0].inst.hex == tracepoint->second) + { + ABI_PushRegistersAndAdjustStack({}, 0); + ABI_CallFunctionPC(RecordPPCEvent, &m_system, em_address); + ABI_PopRegistersAndAdjustStack({}, 0); + } + // Used to get a trace of the last few blocks before a crash, sometimes VERY useful if (m_im_here_debug) { diff --git a/Source/Core/Core/PowerPC/Jit64/JitAsm.cpp b/Source/Core/Core/PowerPC/Jit64/JitAsm.cpp index 7594d0e90e..97474ee273 100644 --- a/Source/Core/Core/PowerPC/Jit64/JitAsm.cpp +++ b/Source/Core/Core/PowerPC/Jit64/JitAsm.cpp @@ -262,6 +262,10 @@ void Jit64AsmRoutineManager::GenerateCommon() GenFres(); mfcr = AlignCode4(); GenMfcr(); + dcache32_read_hit_dbat = AlignCode4(); + GenDCache32Hit(false); + dcache32_write_hit_dbat = AlignCode4(); + GenDCache32Hit(true); cdts = AlignCode4(); GenConvertDoubleToSingle(); fmadds_eft = AlignCode4(); diff --git a/Source/Core/Core/PowerPC/Jit64Common/EmuCodeBlock.cpp b/Source/Core/Core/PowerPC/Jit64Common/EmuCodeBlock.cpp index 1019aa7d0a..ea8815e52e 100644 --- a/Source/Core/Core/PowerPC/Jit64Common/EmuCodeBlock.cpp +++ b/Source/Core/Core/PowerPC/Jit64Common/EmuCodeBlock.cpp @@ -376,9 +376,53 @@ void EmuCodeBlock::SafeLoadToReg(X64Reg reg_value, const Gen::OpArg& opAddress, LEA(32, RSCRATCH, MDisp(opAddress.GetSimpleReg(), offset)); } - FixupBranch exit; const bool dr_set = (flags & SAFE_LOADSTORE_DR_ON) || (m_jit.m_ppc_state.feature_flags & FEATURE_FLAG_MSR_DR); + FixupBranch accurate_dcache_done; + const bool accurate_dcache_hit_path = + !force_slow_access && dr_set && accessSize == 32 && !m_jit.jo.memcheck && + m_jit.m_ppc_state.m_enable_dcache && + (reg_addr != ABI_RETURN || (offset && opAddress.GetSimpleReg() != ABI_RETURN)); + if (accurate_dcache_hit_path) + { + BitSet32 fast_registers_in_use = registersInUse; + if (reg_addr != ABI_RETURN) + fast_registers_in_use[reg_addr] = true; + else + fast_registers_in_use[opAddress.GetSimpleReg()] = true; + + // Preserve scratch registers that remain live, including the original effective address + // needed by a cache-miss fallback. RDX can hold an address or paired-load intermediate even + // though the GPR allocator does not use it. + const bool preserve_rdx = fast_registers_in_use[RSCRATCH2]; + const bool preserve_rcx = fast_registers_in_use[RSCRATCH_EXTRA]; + if (preserve_rdx) + PUSH(RSCRATCH2); + if (preserve_rcx) + PUSH(RSCRATCH_EXTRA); + if (reg_addr != RSCRATCH) + MOV(32, R(RSCRATCH), R(reg_addr)); + CALL(CommonAsmRoutines::dcache32_read_hit_dbat); + if (preserve_rcx) + POP(RSCRATCH_EXTRA); + if (preserve_rdx) + POP(RSCRATCH2); + + TEST(64, R(ABI_RETURN), R(ABI_RETURN)); + const FixupBranch slow = J_CC(CC_Z, Jump::Near); + SUB(64, R(ABI_RETURN), Imm8(1)); + if (reg_value != ABI_RETURN) + MOV(32, R(reg_value), R(ABI_RETURN)); + accurate_dcache_done = J(Jump::Near); + SetJumpTarget(slow); + + // An offset address lives in RAX, which is also the ABI return register. Recreate it only on + // the miss path before calling the full helper. + if (reg_addr == ABI_RETURN && offset) + LEA(32, RSCRATCH, MDisp(opAddress.GetSimpleReg(), offset)); + } + + FixupBranch exit; const bool fast_check_address = !force_slow_access && dr_set && m_jit.jo.fastmem_arena && !m_jit.m_ppc_state.m_enable_dcache; if (fast_check_address) @@ -438,6 +482,9 @@ void EmuCodeBlock::SafeLoadToReg(X64Reg reg_value, const Gen::OpArg& opAddress, } SetJumpTarget(exit); } + + if (accurate_dcache_hit_path) + SetJumpTarget(accurate_dcache_done); } void EmuCodeBlock::SafeLoadToRegImmediate(X64Reg reg_value, u32 address, int accessSize, @@ -533,12 +580,15 @@ void EmuCodeBlock::SafeWriteRegToReg(OpArg reg_value, X64Reg reg_addr, int acces return; } + const X64Reg original_reg_addr = reg_addr; + bool address_in_return_register = false; if (offset) { if (flags & SAFE_LOADSTORE_CLOBBER_RSCRATCH_INSTEAD_OF_ADDR) { LEA(32, RSCRATCH, MDisp(reg_addr, (u32)offset)); reg_addr = RSCRATCH; + address_in_return_register = true; } else { @@ -546,9 +596,53 @@ void EmuCodeBlock::SafeWriteRegToReg(OpArg reg_value, X64Reg reg_addr, int acces } } - FixupBranch exit; const bool dr_set = (flags & SAFE_LOADSTORE_DR_ON) || (m_jit.m_ppc_state.feature_flags & FEATURE_FLAG_MSR_DR); + FixupBranch accurate_dcache_done; + const bool accurate_dcache_hit_path = + !force_slow_access && dr_set && accessSize == 32 && swap && !m_jit.jo.memcheck && + m_jit.m_ppc_state.m_enable_dcache && + (reg_addr != ABI_RETURN || (address_in_return_register && original_reg_addr != ABI_RETURN)) && + (!reg_value.IsSimpleReg() || reg_value.GetSimpleReg() != ABI_RETURN); + if (accurate_dcache_hit_path) + { + BitSet32 fast_registers_in_use = registersInUse; + if (reg_addr != ABI_RETURN) + fast_registers_in_use[reg_addr] = true; + else + fast_registers_in_use[original_reg_addr] = true; + if (reg_value.IsSimpleReg()) + fast_registers_in_use[reg_value.GetSimpleReg()] = true; + + // The value setup itself overwrites RDX, so save it before loading the call arguments. + // On a miss, the full MMU helper must see the original address and value registers. + const bool preserve_rdx = fast_registers_in_use[RSCRATCH2]; + const bool preserve_rcx = fast_registers_in_use[RSCRATCH_EXTRA]; + if (preserve_rdx) + PUSH(RSCRATCH2); + if (preserve_rcx) + PUSH(RSCRATCH_EXTRA); + // Move the address first because it is allowed to arrive in RDX, which is also the native + // routine's value input. + if (reg_addr != RSCRATCH) + MOV(32, R(RSCRATCH), R(reg_addr)); + if (reg_value.IsImm()) + MOV(32, R(RSCRATCH2), reg_value); + else if (reg_value.GetSimpleReg() != RSCRATCH2) + MOV(32, R(RSCRATCH2), reg_value); + CALL(CommonAsmRoutines::dcache32_write_hit_dbat); + if (preserve_rcx) + POP(RSCRATCH_EXTRA); + if (preserve_rdx) + POP(RSCRATCH2); + + TEST(8, R(ABI_RETURN), R(ABI_RETURN)); + accurate_dcache_done = J_CC(CC_NZ, Jump::Near); + if (address_in_return_register) + LEA(32, RSCRATCH, MDisp(original_reg_addr, (u32)offset)); + } + + FixupBranch exit; const bool fast_check_address = !force_slow_access && dr_set && m_jit.jo.fastmem_arena && !m_jit.m_ppc_state.m_enable_dcache; if (fast_check_address) @@ -615,6 +709,9 @@ void EmuCodeBlock::SafeWriteRegToReg(OpArg reg_value, X64Reg reg_addr, int acces } SetJumpTarget(exit); } + + if (accurate_dcache_hit_path) + SetJumpTarget(accurate_dcache_done); } void EmuCodeBlock::SafeWriteRegToReg(Gen::X64Reg reg_value, Gen::X64Reg reg_addr, int accessSize, diff --git a/Source/Core/Core/PowerPC/Jit64Common/Jit64AsmCommon.cpp b/Source/Core/Core/PowerPC/Jit64Common/Jit64AsmCommon.cpp index 4e7e6fe076..f128d7dc0e 100644 --- a/Source/Core/Core/PowerPC/Jit64Common/Jit64AsmCommon.cpp +++ b/Source/Core/Core/PowerPC/Jit64Common/Jit64AsmCommon.cpp @@ -14,8 +14,12 @@ #include "Common/x64ABI.h" #include "Common/x64Emitter.h" #include "Core/PowerPC/Gekko.h" +#include "Core/PowerPC/Jit64/Jit.h" #include "Core/PowerPC/Jit64Common/Jit64Constants.h" #include "Core/PowerPC/Jit64Common/Jit64PowerPCState.h" +#include "Core/PowerPC/MMU.h" +#include "Core/PowerPC/PPCCache.h" +#include "Core/System.h" #define QUANTIZED_REGS_TO_SAVE \ (ABI_ALL_CALLER_SAVED & ~BitSet32{RSCRATCH, RSCRATCH2, RSCRATCH_EXTRA, XMM0 + 16, XMM1 + 16}) @@ -24,11 +28,166 @@ using namespace Gen; +const u8* CommonAsmRoutines::dcache32_read_hit_dbat = nullptr; +const u8* CommonAsmRoutines::dcache32_write_hit_dbat = nullptr; + alignas(16) static const __m128i double_fraction = _mm_set_epi64x(0, 0x000fffffffffffff); alignas(16) static const __m128i double_explicit_top_bit = _mm_set_epi64x(0, 0x0010000000000000); alignas(16) static const __m128i double_top_two_bits = _mm_set_epi64x(0, 0xc000000000000000); alignas(16) static const __m128i double_bottom_bits = _mm_set_epi64x(0, 0x07ffffffe0000000); +constexpr std::array s_dcache_plru_update = [] { + std::array result{}; + for (u32 old_plru = 0; old_plru < 128; ++old_plru) + { + for (u32 way = 0; way < PowerPC::CACHE_WAYS; ++way) + { + result[old_plru * PowerPC::CACHE_WAYS + way] = + (old_plru & ~PowerPC::Cache::PLRU_MASK[way]) | PowerPC::Cache::PLRU_VALUE[way]; + } + } + return result; +}(); + +void CommonAsmRoutines::GenDCache32Hit(const bool write) +{ + // This is the native equivalent of MMU::TryRead/WriteDCache32ForJit for the BAT-mapped + // MEM1/MEM2 hit case. Keeping it as a shared leaf routine avoids both the large C++ ABI + // register save and duplicating this sequence at every guest load/store. + // + // read: EAX = effective address; RAX = byte-swapped value + 1, or zero on miss + // write: EAX = effective address, EDX = guest value; EAX = one on hit, zero on miss + // Only the three Jit64 scratch registers are clobbered. + const void* const start = GetCodePtr(); + auto& cache = m_jit.m_ppc_state.dCache; + auto& memory = m_jit.m_system.GetMemory(); + ASSERT(!cache.lookup_table.empty()); + ASSERT(!memory.GetEXRAM() || !cache.lookup_table_ex.empty()); + + // A write needs RDX for address translation, so keep its input value below the return address. + if (write) + PUSH(RSCRATCH2); + + MOV(32, R(RSCRATCH2), R(RSCRATCH)); + SHR(32, R(RSCRATCH2), Imm8(PowerPC::BAT_INDEX_SHIFT)); + MOV(64, R(RSCRATCH_EXTRA), ImmPtr(m_jit.m_mmu.GetDBATTable().data())); + MOV(32, R(RSCRATCH2), MComplex(RSCRATCH_EXTRA, RSCRATCH2, SCALE_4, 0)); + MOV(32, R(RSCRATCH_EXTRA), R(RSCRATCH2)); + AND(32, R(RSCRATCH_EXTRA), Imm32(PowerPC::BAT_MAPPED_BIT | PowerPC::BAT_WI_BIT)); + CMP(32, R(RSCRATCH_EXTRA), Imm32(PowerPC::BAT_MAPPED_BIT)); + const FixupBranch invalid_bat = J_CC(CC_NE, Jump::Near); + + AND(32, R(RSCRATCH2), Imm32(PowerPC::BAT_RESULT_MASK)); + AND(32, R(RSCRATCH), Imm32(PowerPC::BAT_PAGE_SIZE - 1)); + OR(32, R(RSCRATCH2), R(RSCRATCH)); + + // Select the lookup table while normalizing the physical address. Cache set and byte offset + // are identical for MEM1 and MEM2 after this normalization. + TEST(32, R(RSCRATCH2), Imm32(0xF8000000)); + const FixupBranch mem1 = J_CC(CC_Z, Jump::Near); + + MOV(32, R(RSCRATCH), R(RSCRATCH2)); + AND(32, R(RSCRATCH), Imm32(0xF0000000)); + CMP(32, R(RSCRATCH), Imm32(0x10000000)); + const FixupBranch not_exram = J_CC(CC_NE, Jump::Near); + AND(32, R(RSCRATCH2), Imm32(0x0FFFFFFF)); + CMP(32, R(RSCRATCH2), Imm32(memory.GetEXRAM() ? memory.GetExRamSizeReal() : 0)); + const FixupBranch exram_out_of_range = J_CC(CC_AE, Jump::Near); + MOV(64, R(RSCRATCH), ImmPtr(cache.lookup_table_ex.data())); + const FixupBranch lookup_ready = J(Jump::Near); + + SetJumpTarget(mem1); + AND(32, R(RSCRATCH2), Imm32(memory.GetRamMask())); + MOV(64, R(RSCRATCH), ImmPtr(cache.lookup_table.data())); + + SetJumpTarget(lookup_ready); + MOV(32, R(RSCRATCH_EXTRA), R(RSCRATCH2)); + SHR(32, R(RSCRATCH_EXTRA), Imm8(5)); + MOVZX(32, 8, RSCRATCH_EXTRA, MComplex(RSCRATCH, RSCRATCH_EXTRA, SCALE_1, 0)); + CMP(32, R(RSCRATCH_EXTRA), Imm32(0xff)); + const FixupBranch cache_miss = J_CC(CC_E, Jump::Near); + + // The generic helper preserves split cache-line transactions. Ordinary aligned 32-bit PPC + // traffic never takes this branch. + MOV(32, R(RSCRATCH), R(RSCRATCH2)); + AND(32, R(RSCRATCH), Imm32(31)); + CMP(32, R(RSCRATCH), Imm32(28)); + const FixupBranch split_access = J_CC(CC_A, Jump::Near); + + // RDX becomes a compact byte index into Cache::data: + // set * (8 ways * 32 bytes) + way * 32 + byte offset. + MOV(32, R(RSCRATCH), R(RSCRATCH2)); + AND(32, R(RSCRATCH), Imm32(0xFE0)); + SHL(32, R(RSCRATCH), Imm8(3)); + AND(32, R(RSCRATCH2), Imm32(31)); + OR(32, R(RSCRATCH2), R(RSCRATCH)); + SHL(32, R(RSCRATCH_EXTRA), Imm8(5)); + OR(32, R(RSCRATCH2), R(RSCRATCH_EXTRA)); + + // Update the exact 8-way pseudo-LRU state. Preserve the compact data index across the lookup. + PUSH(RSCRATCH2); + MOV(32, R(RSCRATCH), R(RSCRATCH2)); + SHR(32, R(RSCRATCH), Imm8(8)); + AND(32, R(RSCRATCH), Imm32(PowerPC::CACHE_SETS - 1)); + MOV(32, R(RSCRATCH_EXTRA), R(RSCRATCH2)); + SHR(32, R(RSCRATCH_EXTRA), Imm8(5)); + AND(32, R(RSCRATCH_EXTRA), Imm32(PowerPC::CACHE_WAYS - 1)); + MOV(64, R(RSCRATCH2), ImmPtr(cache.plru.data())); + MOVZX(32, 8, RSCRATCH2, MComplex(RSCRATCH2, RSCRATCH, SCALE_1, 0)); + SHL(32, R(RSCRATCH2), Imm8(3)); + OR(32, R(RSCRATCH2), R(RSCRATCH_EXTRA)); + MOV(64, R(RSCRATCH_EXTRA), ImmPtr(s_dcache_plru_update.data())); + MOVZX(32, 8, RSCRATCH2, MComplex(RSCRATCH_EXTRA, RSCRATCH2, SCALE_1, 0)); + MOV(64, R(RSCRATCH_EXTRA), ImmPtr(cache.plru.data())); + MOV(8, MComplex(RSCRATCH_EXTRA, RSCRATCH, SCALE_1, 0), R(RSCRATCH2)); + POP(RSCRATCH2); + + if (write) + { + // Restore, byte-swap and store the guest value, then mark the line dirty. + POP(RSCRATCH); + BSWAP(32, RSCRATCH); + MOV(64, R(RSCRATCH_EXTRA), ImmPtr(cache.data.data())); + MOV(32, MComplex(RSCRATCH_EXTRA, RSCRATCH2, SCALE_1, 0), R(RSCRATCH)); + + MOV(32, R(RSCRATCH), R(RSCRATCH2)); + SHR(32, R(RSCRATCH), Imm8(8)); + AND(32, R(RSCRATCH), Imm32(PowerPC::CACHE_SETS - 1)); + MOV(32, R(RSCRATCH_EXTRA), R(RSCRATCH2)); + SHR(32, R(RSCRATCH_EXTRA), Imm8(5)); + AND(32, R(RSCRATCH_EXTRA), Imm32(PowerPC::CACHE_WAYS - 1)); + MOV(32, R(RSCRATCH2), Imm32(1)); + SHL(32, R(RSCRATCH2), R(RSCRATCH_EXTRA)); + MOV(64, R(RSCRATCH_EXTRA), ImmPtr(cache.modified.data())); + OR(8, MComplex(RSCRATCH_EXTRA, RSCRATCH, SCALE_1, 0), R(RSCRATCH2)); + MOV(32, R(RSCRATCH), Imm32(1)); + RET(); + } + else + { + MOV(64, R(RSCRATCH_EXTRA), ImmPtr(cache.data.data())); + MOV(32, R(RSCRATCH), MComplex(RSCRATCH_EXTRA, RSCRATCH2, SCALE_1, 0)); + BSWAP(32, RSCRATCH); + ADD(64, R(RSCRATCH), Imm8(1)); + RET(); + } + + SetJumpTarget(invalid_bat); + SetJumpTarget(not_exram); + SetJumpTarget(exram_out_of_range); + SetJumpTarget(cache_miss); + SetJumpTarget(split_access); + if (write) + POP(RSCRATCH2); + XOR(32, R(RSCRATCH), R(RSCRATCH)); + RET(); + + if (write) + Common::JitRegister::Register(start, GetCodePtr(), "JIT_dcache32_write_hit"); + else + Common::JitRegister::Register(start, GetCodePtr(), "JIT_dcache32_read_hit"); +} + // Since the following float conversion functions are used in non-arithmetic PPC float // instructions, they must convert floats bitexact and never flush denormals to zero or turn SNaNs // into QNaNs. This means we can't use CVTSS2SD/CVTSD2SS. diff --git a/Source/Core/Core/PowerPC/Jit64Common/Jit64AsmCommon.h b/Source/Core/Core/PowerPC/Jit64Common/Jit64AsmCommon.h index 8d60b005b3..94134e7cb2 100644 --- a/Source/Core/Core/PowerPC/Jit64Common/Jit64AsmCommon.h +++ b/Source/Core/Core/PowerPC/Jit64Common/Jit64AsmCommon.h @@ -27,9 +27,16 @@ class CommonAsmRoutines : public CommonAsmRoutinesBase, public QuantizedMemoryRo { public: explicit CommonAsmRoutines(Jit64& jit) : QuantizedMemoryRoutines(jit) {} + + // Accurate Broadway D-cache leaf routines. Static storage intentionally keeps the layout of + // Jit64AsmRoutineManager (and therefore Jit64) unchanged. + static const u8* dcache32_read_hit_dbat; + static const u8* dcache32_write_hit_dbat; + void GenFrsqrte(); void GenFres(); void GenMfcr(); + void GenDCache32Hit(bool write); protected: void GenConvertDoubleToSingle(); diff --git a/Source/Core/Core/PowerPC/MMU.cpp b/Source/Core/Core/PowerPC/MMU.cpp index 5b5faada0c..73a545694c 100644 --- a/Source/Core/Core/PowerPC/MMU.cpp +++ b/Source/Core/Core/PowerPC/MMU.cpp @@ -50,6 +50,7 @@ #include "Core/HW/MMIO.h" #include "Core/HW/Memmap.h" #include "Core/HW/ProcessorInterface.h" +#include "Core/HW/VideoInterface.h" #include "Core/HW/WII_IPC.h" #include "Core/IOS/Starlet/Starlet.h" #include "Core/IOS/Starlet/StarletMemory.h" @@ -250,6 +251,17 @@ T MMU::ReadFromHardware(u32 em_address) } } + // The Wii Menu polls VI_VERTICAL_BEAM_POSITION in a very tight loop while synchronizing its + // startup screens. Keep the exact live VI value, but avoid routing every 16-bit read through the + // generic MMIO mapping, type-erased handler and std::function layers. Address translation and + // all timing updates still happen normally before this point. + if constexpr (flag == XCheckTLBFlag::Read && sizeof(T) == sizeof(u16)) + { + constexpr u32 vi_vertical_beam_position = 0x0c002000 | VideoInterface::VI_VERTICAL_BEAM_POSITION; + if (em_address == vi_vertical_beam_position) + return static_cast(m_system.GetVideoInterface().GetVerticalBeamPosition()); + } + if (flag == XCheckTLBFlag::Read && (em_address & 0xF8000000) == 0x08000000) { if (em_address < 0x0c000000) @@ -821,6 +833,213 @@ void MMU::Write(const u64 var, const u32 address) WriteToHardware(address + sizeof(u32), static_cast(var), 4); } +template +T MMU::ReadForJit(const u32 address) +{ + if (!m_ppc_state.m_enable_dcache || m_power_pc.GetMemChecks().HasAny() || + (address & HW_PAGE_MASK) > HW_PAGE_SIZE - sizeof(T)) + { + return Read(address); + } + + u32 physical_address = address; + if (m_ppc_state.msr.DR) + { + const u32 bat_result = m_dbat_table[address >> BAT_INDEX_SHIFT]; + if ((bat_result & BAT_MAPPED_BIT) == 0 || (bat_result & BAT_WI_BIT) != 0) + return Read(address); + + physical_address = + (bat_result & BAT_RESULT_MASK) | (address & (BAT_PAGE_SIZE - 1)); + } + + if (m_memory.GetRAM() && (physical_address & 0xF8000000) == 0) + { + physical_address &= m_memory.GetRamMask(); + const T value = m_ppc_state.dCache.ReadMainMemoryValue( + m_memory, physical_address, HID0(m_ppc_state).DLOCK); + return Common::FromBigEndian(value); + } + + if (m_memory.GetEXRAM() && (physical_address >> 28) == 1 && + (physical_address & 0x0FFFFFFF) < m_memory.GetExRamSizeReal()) + { + physical_address = (physical_address & 0x0FFFFFFF) + CACHE_EXRAM_BIT; + const T value = m_ppc_state.dCache.ReadMainMemoryValue( + m_memory, physical_address, HID0(m_ppc_state).DLOCK); + return Common::FromBigEndian(value); + } + + return Read(address); +} +template u8 MMU::ReadForJit(u32 address); +template u16 MMU::ReadForJit(u32 address); +template u32 MMU::ReadForJit(u32 address); +template u64 MMU::ReadForJit(u32 address); + +template +void MMU::WriteForJit(const Common::MakeAtLeastU32 var, const u32 address) +{ + // A 64-bit Broadway store is represented by two 32-bit bus writes in the generic path. Keep + // that uncommon operation there; byte, halfword and word animation/state stores use this path. + if constexpr (sizeof(T) > sizeof(u32)) + { + Write(var, address); + return; + } + + if (!m_ppc_state.m_enable_dcache || m_power_pc.GetMemChecks().HasAny() || + (address & HW_PAGE_MASK) > HW_PAGE_SIZE - sizeof(T)) + { + Write(var, address); + return; + } + + u32 physical_address = address; + if (m_ppc_state.msr.DR) + { + const u32 bat_result = m_dbat_table[address >> BAT_INDEX_SHIFT]; + if ((bat_result & BAT_MAPPED_BIT) == 0 || (bat_result & BAT_WI_BIT) != 0) + { + Write(var, address); + return; + } + + physical_address = + (bat_result & BAT_RESULT_MASK) | (address & (BAT_PAGE_SIZE - 1)); + } + + const T swapped_value = Common::FromBigEndian(static_cast(var)); + if (m_memory.GetRAM() && (physical_address & 0xF8000000) == 0) + { + physical_address &= m_memory.GetRamMask(); + m_ppc_state.dCache.WriteMainMemoryValue( + m_memory, physical_address, swapped_value, HID0(m_ppc_state).DLOCK); + return; + } + + if (m_memory.GetEXRAM() && (physical_address >> 28) == 1 && + (physical_address & 0x0FFFFFFF) < m_memory.GetExRamSizeReal()) + { + physical_address = (physical_address & 0x0FFFFFFF) + CACHE_EXRAM_BIT; + m_ppc_state.dCache.WriteMainMemoryValue( + m_memory, physical_address, swapped_value, HID0(m_ppc_state).DLOCK); + return; + } + + Write(var, address); +} +template void MMU::WriteForJit(u32 var, u32 address); +template void MMU::WriteForJit(u32 var, u32 address); +template void MMU::WriteForJit(u32 var, u32 address); +template void MMU::WriteForJit(u64 var, u32 address); + +u64 MMU::TryReadDCache32ForJit(const u32 address) +{ + // The generic helper handles the rare split transaction precisely. + if ((address & HW_PAGE_MASK) > HW_PAGE_SIZE - sizeof(u32)) + return 0; + + u32 physical_address = address; + if (m_ppc_state.msr.DR) + { + const u32 bat_result = m_dbat_table[address >> BAT_INDEX_SHIFT]; + if ((bat_result & (BAT_MAPPED_BIT | BAT_WI_BIT)) != BAT_MAPPED_BIT) + return 0; + physical_address = + (bat_result & BAT_RESULT_MASK) | (address & (BAT_PAGE_SIZE - 1)); + } + + bool exram = false; + u32 lookup_index; + if (m_memory.GetRAM() && (physical_address & 0xF8000000) == 0) + { + physical_address &= m_memory.GetRamMask(); + lookup_index = physical_address >> 5; + } + else if (m_memory.GetEXRAM() && (physical_address >> 28) == 1 && + (physical_address & 0x0FFFFFFF) < m_memory.GetExRamSizeReal()) + { + exram = true; + physical_address = (physical_address & 0x0FFFFFFF) + CACHE_EXRAM_BIT; + lookup_index = (physical_address & 0x0FFFFFFF) >> 5; + } + else + { + return 0; + } + + if ((physical_address & 31) > 32 - sizeof(u32)) + return 0; + + Cache& cache = m_ppc_state.dCache; + const u32 line_address = physical_address & ~31U; + const u32 set = (line_address >> 5) & 0x7f; + const u32 way = exram ? cache.lookup_table_ex[lookup_index] : cache.lookup_table[lookup_index]; + if (way == 0xff) + return 0; + + cache.plru[set] = (cache.plru[set] & ~Cache::PLRU_MASK[way]) | Cache::PLRU_VALUE[way]; + u32 value; + std::memcpy(&value, + reinterpret_cast(cache.data[set][way].data()) + + (physical_address & 31), + sizeof(value)); + return static_cast(Common::swap32(value)) + 1; +} + +bool MMU::TryWriteDCache32ForJit(const u32 var, const u32 address) +{ + if ((address & HW_PAGE_MASK) > HW_PAGE_SIZE - sizeof(u32)) + return false; + + u32 physical_address = address; + if (m_ppc_state.msr.DR) + { + const u32 bat_result = m_dbat_table[address >> BAT_INDEX_SHIFT]; + if ((bat_result & (BAT_MAPPED_BIT | BAT_WI_BIT)) != BAT_MAPPED_BIT) + return false; + physical_address = + (bat_result & BAT_RESULT_MASK) | (address & (BAT_PAGE_SIZE - 1)); + } + + bool exram = false; + u32 lookup_index; + if (m_memory.GetRAM() && (physical_address & 0xF8000000) == 0) + { + physical_address &= m_memory.GetRamMask(); + lookup_index = physical_address >> 5; + } + else if (m_memory.GetEXRAM() && (physical_address >> 28) == 1 && + (physical_address & 0x0FFFFFFF) < m_memory.GetExRamSizeReal()) + { + exram = true; + physical_address = (physical_address & 0x0FFFFFFF) + CACHE_EXRAM_BIT; + lookup_index = (physical_address & 0x0FFFFFFF) >> 5; + } + else + { + return false; + } + + if ((physical_address & 31) > 32 - sizeof(u32)) + return false; + + Cache& cache = m_ppc_state.dCache; + const u32 line_address = physical_address & ~31U; + const u32 set = (line_address >> 5) & 0x7f; + const u32 way = exram ? cache.lookup_table_ex[lookup_index] : cache.lookup_table[lookup_index]; + if (way == 0xff) + return false; + + cache.plru[set] = (cache.plru[set] & ~Cache::PLRU_MASK[way]) | Cache::PLRU_VALUE[way]; + const u32 swapped_value = Common::swap32(var); + std::memcpy(reinterpret_cast(cache.data[set][way].data()) + (physical_address & 31), + &swapped_value, sizeof(swapped_value)); + cache.modified[set] |= 1U << way; + return true; +} + void MMU::Write_U16_Swap(const u32 var, const u32 address) { Write((var & 0xFFFF0000) | Common::swap16(static_cast(var)), address); @@ -2183,10 +2402,18 @@ void ClearDCacheLineFromJit(MMU& mmu, u32 address) { mmu.ClearDCacheLine(address); } +u64 TryReadDCache32FromJit(MMU& mmu, u32 address) +{ + return mmu.TryReadDCache32ForJit(address); +} +bool TryWriteDCache32FromJit(MMU& mmu, u32 var, u32 address) +{ + return mmu.TryWriteDCache32ForJit(var, address); +} template Common::MakeAtLeastU32 ReadFromJit(MMU& mmu, u32 address) { - return mmu.Read(address); + return mmu.ReadForJit(address); } template u32 ReadFromJit(MMU& mmu, u32 address); template u32 ReadFromJit(MMU& mmu, u32 address); @@ -2196,7 +2423,7 @@ template u64 ReadFromJit(MMU& mmu, u32 address); template void WriteFromJit(MMU& mmu, Common::MakeAtLeastU32 var, u32 address) { - mmu.Write(var, address); + mmu.WriteForJit(var, address); } template void WriteFromJit(MMU& mmu, u32 var, u32 address); template void WriteFromJit(MMU& mmu, u32 var, u32 address); diff --git a/Source/Core/Core/PowerPC/MMU.h b/Source/Core/Core/PowerPC/MMU.h index 5e4c765102..7787b5a822 100644 --- a/Source/Core/Core/PowerPC/MMU.h +++ b/Source/Core/Core/PowerPC/MMU.h @@ -233,6 +233,21 @@ public: template void Write(Common::MakeAtLeastU32 var, u32 address); + // JIT memory helpers. When accurate D-cache emulation is enabled, ordinary fastmem cannot be + // used because Broadway loads and stores must update the emulated cache rather than backing RAM. + // These helpers retain those exact semantics while bypassing the generic hardware dispatcher for + // the overwhelmingly common BAT-mapped MEM1/MEM2 case. + template + T ReadForJit(u32 address); + template + void WriteForJit(Common::MakeAtLeastU32 var, u32 address); + + // Leaf fast paths used by Jit64 when accurate D-cache emulation is active. A zero/false result + // means that the caller must use ReadForJit/WriteForJit (cache miss, MMIO, TLB, WI, etc.). Reads + // encode a successful 32-bit value as value+1 in 64 bits so every guest value remains representable. + u64 TryReadDCache32ForJit(u32 address); + bool TryWriteDCache32ForJit(u32 var, u32 address); + void Write_U16_Swap(u32 var, u32 address); void Write_U32_Swap(u32 var, u32 address); void Write_U64_Swap(u64 var, u32 address); @@ -374,6 +389,8 @@ private: }; void ClearDCacheLineFromJit(MMU& mmu, u32 address); +u64 TryReadDCache32FromJit(MMU& mmu, u32 address); +bool TryWriteDCache32FromJit(MMU& mmu, u32 var, u32 address); template // Returns zero-extended value Common::MakeAtLeastU32 ReadFromJit(MMU& mmu, u32 address); diff --git a/Source/Core/Core/PowerPC/PPCAnalyst.cpp b/Source/Core/Core/PowerPC/PPCAnalyst.cpp index 386e349e3b..7266cf9c84 100644 --- a/Source/Core/Core/PowerPC/PPCAnalyst.cpp +++ b/Source/Core/Core/PowerPC/PPCAnalyst.cpp @@ -734,10 +734,11 @@ void PPCAnalyzer::SetInstructionStats(CodeBlock* block, CodeOp* code, } } -bool PPCAnalyzer::IsBusyWaitLoop(CodeBlock* block, CodeOp* code, size_t instructions) const +bool PPCAnalyzer::IsBusyWaitLoop(CodeOp* code, size_t loop_start, size_t branch_index) const { // Very basic algorithm to detect busy wait loops: - // * It loops to itself and does not contain any other branches. + // * It loops to the first instruction in the candidate range and does not contain any other + // branches. // * It does not write to memory. // * It only reads from registers it wrote to earlier in the loop, or it // does not write to these registers. @@ -748,14 +749,15 @@ bool PPCAnalyzer::IsBusyWaitLoop(CodeBlock* block, CodeOp* code, size_t instruct // don't detect these at the moment. std::bitset<32> write_disallowed_regs; std::bitset<32> written_regs; - for (size_t i = 0; i <= instructions; ++i) + for (size_t i = loop_start; i <= branch_index; ++i) { if (code[i].opinfo->type == OpType::Branch) { if (code[i].branchUsesCtr) return false; - if (code[i].branchTo == block->m_address && i == instructions) + if (code[i].branchTo == code[loop_start].address && i == branch_index) return true; + return false; } // A `nop` is actually a `ori r0, r0, 0`, which would violate the rules (unless `r0` was written // earlier). @@ -763,6 +765,12 @@ bool PPCAnalyzer::IsBusyWaitLoop(CodeBlock* block, CodeOp* code, size_t instruct { continue; } + // The JIT implements sync as a host no-op. It is commonly placed directly after an MMIO load + // in hardware polling loops, and does not make a load-only loop unsafe to idle. + else if (code[i].inst.OPCD == 31 && code[i].inst.SUBOP10 == 598) + { + continue; + } else if (code[i].opinfo->type != OpType::Integer && code[i].opinfo->type != OpType::Load) { // In the future, some subsets of other instruction types might get @@ -946,8 +954,22 @@ u32 PPCAnalyzer::Analyze(u32 address, CodeBlock* block, CodeBuffer* buffer, } } - code[i].branchIsIdleLoop = - code[i].branchTo == block->m_address && IsBusyWaitLoop(block, code, i); + code[i].branchIsIdleLoop = false; + if (code[i].opinfo->type == OpType::Branch) + { + // Branch following can inline a small polling function into its caller. Such a loop branches + // to an instruction in the middle of the analyzed block rather than block->m_address, so the + // old detector missed it. Locate the innermost matching instruction and validate only that + // loop range. + for (size_t candidate = i + 1; candidate-- > 0;) + { + if (code[candidate].address == code[i].branchTo) + { + code[i].branchIsIdleLoop = IsBusyWaitLoop(code, candidate, i); + break; + } + } + } if (follow && numFollows < BRANCH_FOLLOWING_THRESHOLD) { diff --git a/Source/Core/Core/PowerPC/PPCAnalyst.h b/Source/Core/Core/PowerPC/PPCAnalyst.h index f9081bae93..9dd799a71c 100644 --- a/Source/Core/Core/PowerPC/PPCAnalyst.h +++ b/Source/Core/Core/PowerPC/PPCAnalyst.h @@ -199,7 +199,7 @@ private: ReorderType type) const; void ReorderInstructions(u32 instructions, CodeOp* code) const; void SetInstructionStats(CodeBlock* block, CodeOp* code, const GekkoOPInfo* opinfo) const; - bool IsBusyWaitLoop(CodeBlock* block, CodeOp* code, size_t instructions) const; + bool IsBusyWaitLoop(CodeOp* code, size_t loop_start, size_t branch_index) const; // Options u32 m_options = 0; diff --git a/Source/Core/Core/PowerPC/PowerPC.cpp b/Source/Core/Core/PowerPC/PowerPC.cpp index 564d74d1f6..31098536e6 100644 --- a/Source/Core/Core/PowerPC/PowerPC.cpp +++ b/Source/Core/Core/PowerPC/PowerPC.cpp @@ -275,10 +275,14 @@ void PowerPCManager::Init(CPUCore cpu_core) Reset(); - InitializeCPUCore(cpu_core); auto& memory = m_system.GetMemory(); m_ppc_state.iCache.Init(memory); m_ppc_state.dCache.Init(memory); + + // JIT common routines may embed stable pointers into the cache lookup tables. Allocate those + // tables before generating a CPU core, rather than leaving the JIT with the empty vectors from + // PowerPCState construction. Reset and state loads only refill these allocations afterwards. + InitializeCPUCore(cpu_core); } void PowerPCManager::Reset() diff --git a/Source/Core/Core/State.cpp b/Source/Core/Core/State.cpp index 2dcd2cae2b..0aa77abc3c 100644 --- a/Source/Core/Core/State.cpp +++ b/Source/Core/Core/State.cpp @@ -97,7 +97,7 @@ struct CompressAndDumpStateArgs static Common::WorkQueueThreadSP s_compress_and_dump_thread; // Don't forget to increase this after doing changes on the savestate system -constexpr u32 STATE_VERSION = 192; // Last changed in PR 14646 +constexpr u32 STATE_VERSION = 193; // Starlet HW_TIMER counter offset. // Increase this if the StateExtendedHeader definition changes constexpr u32 EXTENDED_HEADER_VERSION = 1; // Last changed in PR 12217 diff --git a/Source/UnitTests/Core/CMakeLists.txt b/Source/UnitTests/Core/CMakeLists.txt index 4fbc7492d0..2817fb88ad 100644 --- a/Source/UnitTests/Core/CMakeLists.txt +++ b/Source/UnitTests/Core/CMakeLists.txt @@ -26,6 +26,7 @@ if(_M_X86_64) PowerPC/DivUtilsTest.cpp PowerPC/PageTableHostMappingTest.cpp PowerPC/Jit64Common/ConvertDoubleToSingle.cpp + PowerPC/Jit64Common/DCache.cpp PowerPC/Jit64Common/Fres.cpp PowerPC/Jit64Common/Frsqrte.cpp ) diff --git a/Source/UnitTests/Core/IOS/Starlet/ARMCoreTest.cpp b/Source/UnitTests/Core/IOS/Starlet/ARMCoreTest.cpp index da1b7dd009..97d76c49a1 100644 --- a/Source/UnitTests/Core/IOS/Starlet/ARMCoreTest.cpp +++ b/Source/UnitTests/Core/IOS/Starlet/ARMCoreTest.cpp @@ -10,6 +10,7 @@ #include +#include "Common/ChunkFile.h" #include "Common/CommonTypes.h" #include "Core/Core.h" #include "Core/HW/WII_IPC.h" @@ -171,6 +172,14 @@ public: return m_sram_fastmem_enabled ? &m_sram_split_mode : nullptr; } + const u8* GetDirectMemoryPointer(u32 address, u32 size) const override + { + const size_t offset = ToOffset(address); + if (size == 0 || offset > m_memory.size() || size > m_memory.size() - offset) + return nullptr; + return m_memory.data() + offset; + } + void SetIdlePollSafe(bool safe) { m_idle_poll_safe = safe; } void SetSliceStablePollAddress(u32 address) { @@ -376,7 +385,7 @@ TEST(StarletTimer, ZeroDelayAlarmMatchesImmediatelyAndUsesIRQW1C) memory.Write8(address + 3, static_cast(value)); }; - memory.AdvanceCycles(405); + memory.AdvanceCycles(32 * 128); ASSERT_EQ(read_word(timer), 32u); write_word(alarm, read_word(timer)); EXPECT_EQ(system.GetWiiIPC().ReadStarletRegister(0x38) & INT_CAUSE_TIMER, INT_CAUSE_TIMER); @@ -388,12 +397,155 @@ TEST(StarletTimer, ZeroDelayAlarmMatchesImmediatelyAndUsesIRQW1C) write_word(arm_irq_flag, INT_CAUSE_TIMER); EXPECT_EQ(system.GetWiiIPC().ReadStarletRegister(0x38) & INT_CAUSE_TIMER, 0u); - memory.AdvanceCycles(404); + memory.AdvanceCycles(32 * 128 - 1); EXPECT_EQ(system.GetWiiIPC().ReadStarletRegister(0x38) & INT_CAUSE_TIMER, 0u); memory.AdvanceCycles(1); EXPECT_EQ(system.GetWiiIPC().ReadStarletRegister(0x38) & INT_CAUSE_TIMER, INT_CAUSE_TIMER); } +TEST(StarletRegisters, WideTimerAndInterruptAccessesMatchHardwareSemantics) +{ + constexpr u32 hardware_base = 0x0d800000; + constexpr u32 timer = hardware_base + 0x10; + constexpr u32 arm_irq_flag = hardware_base + 0x38; + constexpr u32 arm_irq_mask = hardware_base + 0x3c; + Core::DeclareAsCPUThread(); + auto& system = Core::System::GetInstance(); + system.GetWiiIPC().Reset(); + StarletMemory memory(system); + memory.Reset(); + + memory.AdvanceCycles(32 * 128); + EXPECT_EQ(memory.Read32(timer), 32u); + memory.Write32(timer, 64); + EXPECT_EQ(memory.Read32(timer), 64u); + + system.GetWiiIPC().SetStarletInterrupt(INT_CAUSE_TIMER, true); + EXPECT_EQ(memory.Read32(arm_irq_flag) & INT_CAUSE_TIMER, INT_CAUSE_TIMER); + memory.Write32(arm_irq_flag, INT_CAUSE_TIMER); + EXPECT_EQ(memory.Read32(arm_irq_flag) & INT_CAUSE_TIMER, 0u); + + memory.Write32(arm_irq_mask, 0x800619ef); + EXPECT_EQ(memory.Read32(arm_irq_mask), 0x800619efu); +} + +TEST(StarletTimer, RunsAtOneTickPer128ARMCycles) +{ + constexpr u32 timer = 0x0d800010; + Core::DeclareAsCPUThread(); + auto& system = Core::System::GetInstance(); + system.GetWiiIPC().Reset(); + StarletMemory memory(system); + memory.Reset(); + + memory.AdvanceCycles(127); + EXPECT_EQ(memory.Read32(timer), 0u); + memory.AdvanceCycles(1); + EXPECT_EQ(memory.Read32(timer), 1u); + memory.AdvanceCycles(243'000'000 - 128); + EXPECT_EQ(memory.Read32(timer), 1'898'437u); + memory.AdvanceCycles(243'000'000); + EXPECT_EQ(memory.Read32(timer), 3'796'875u); +} + +TEST(StarletTimer, SchedulerSlicePartitionDoesNotChangeClock) +{ + Core::DeclareAsCPUThread(); + auto& system = Core::System::GetInstance(); + StarletMemory memory(system); + memory.Reset(); + + constexpr u64 total_cycles = 243'000'000; + // Active, IPC, and idle scheduler slices must use the same clock, with no + // fractional timer ticks lost at the end of a slice. + for (const u64 slice : {256u, 4096u, 24300u}) + { + memory.Reset(); + for (u64 elapsed = 0; elapsed < total_cycles;) + { + const u64 step = std::min(slice, total_cycles - elapsed); + memory.AdvanceCycles(step); + elapsed += step; + } + EXPECT_EQ(memory.Read32(0x0d800010), 1'898'437u) << "slice=" << slice; + EXPECT_EQ(memory.GetCycles(), total_cycles); + } +} + +TEST(StarletTimer, CounterWritesDoNotRewindPeripheralClock) +{ + constexpr u32 timer = 0x0d800010; + Core::DeclareAsCPUThread(); + auto& system = Core::System::GetInstance(); + StarletMemory memory(system); + memory.Reset(); + memory.AdvanceCycles(1025); + + for (const u32 value : {1u, 0xffffffffu, 0u, 0x12345678u}) + { + memory.Write32(timer, value); + EXPECT_EQ(memory.Read32(timer), value); + EXPECT_EQ(memory.GetCycles(), 1025u); + } + // Reprogramming HW_TIMER leaves the free-running /128 clock phase intact. + memory.AdvanceCycles(126); + EXPECT_EQ(memory.Read32(timer), 0x12345678u); + memory.AdvanceCycles(1); + EXPECT_EQ(memory.Read32(timer), 0x12345679u); +} + +TEST(StarletTimer, ByteAssembledCounterWritesMatchWideWrites) +{ + constexpr u32 timer = 0x0d800010; + Core::DeclareAsCPUThread(); + auto& system = Core::System::GetInstance(); + StarletMemory memory(system); + memory.Reset(); + memory.AdvanceCycles(1280); + memory.Write8(timer, 0x12); + memory.Write8(timer + 1, 0x34); + memory.Write8(timer + 2, 0x56); + memory.Write8(timer + 3, 0x78); + EXPECT_EQ(memory.Read32(timer), 0x12345678u); + EXPECT_EQ(memory.GetCycles(), 1280u); + memory.AdvanceCycles(128); + EXPECT_EQ(memory.Read32(timer), 0x12345679u); +} + +TEST(StarletTimer, AlarmFiresAcrossCounterWrap) +{ + constexpr u32 timer = 0x0d800010; + constexpr u32 alarm = 0x0d800014; + Core::DeclareAsCPUThread(); + auto& system = Core::System::GetInstance(); + system.GetWiiIPC().Reset(); + StarletMemory memory(system); + memory.Reset(); + memory.Write32(timer, 0xfffffffe); + memory.Write32(alarm, 1); + memory.AdvanceCycles(3 * 128 - 1); + EXPECT_EQ(memory.Read32(timer), 0u); + EXPECT_EQ(system.GetWiiIPC().ReadStarletRegister(0x38) & INT_CAUSE_TIMER, 0u); + memory.AdvanceCycles(1); + EXPECT_EQ(memory.Read32(timer), 1u); + EXPECT_EQ(system.GetWiiIPC().ReadStarletRegister(0x38) & INT_CAUSE_TIMER, INT_CAUSE_TIMER); +} + +TEST(StarletTimer, ResetClearsCounterOffset) +{ + Core::DeclareAsCPUThread(); + auto& system = Core::System::GetInstance(); + StarletMemory memory(system); + memory.Reset(); + memory.AdvanceCycles(512); + memory.Write32(0x0d800010, 0x12345678); + memory.Reset(); + EXPECT_EQ(memory.GetCycles(), 0u); + EXPECT_EQ(memory.Read32(0x0d800010), 0u); + memory.AdvanceCycles(128); + EXPECT_EQ(memory.Read32(0x0d800010), 1u); +} + TEST(StarletNAND, HardwareResetPreservesProgrammedFlash) { Core::DeclareAsCPUThread(); @@ -426,6 +578,36 @@ TEST(StarletNAND, HardwareResetPreservesProgrammedFlash) EXPECT_EQ(memory.Read32(sram), 0x12345678u); } +TEST(StarletTimer, StateRoundTripPreservesOffsetAndDividerPhase) +{ + constexpr u32 timer = 0x0d800010; + Core::DeclareAsCPUThread(); + auto& system = Core::System::GetInstance(); + StarletMemory memory(system); + memory.Reset(); + memory.AdvanceCycles(1025); + memory.Write32(timer, 0x12345678); + + std::vector state_buffer(1024 * 1024); + u8* state_pointer = state_buffer.data(); + PointerWrap writer(&state_pointer, state_buffer.size(), PointerWrap::Mode::Write); + memory.DoState(writer); + ASSERT_TRUE(writer.IsWriteMode()); + const size_t state_size = state_pointer - state_buffer.data(); + + memory.Reset(); + state_pointer = state_buffer.data(); + PointerWrap reader(&state_pointer, state_size, PointerWrap::Mode::Read); + memory.DoState(reader); + ASSERT_TRUE(reader.IsReadMode()); + EXPECT_EQ(memory.GetCycles(), 1025u); + EXPECT_EQ(memory.Read32(timer), 0x12345678u); + memory.AdvanceCycles(126); + EXPECT_EQ(memory.Read32(timer), 0x12345678u); + memory.AdvanceCycles(1); + EXPECT_EQ(memory.Read32(timer), 0x12345679u); +} + TEST(StarletGPIO, InterruptFlagIsWriteOneToClear) { constexpr u32 hardware_base = 0x0d800000; @@ -984,6 +1166,35 @@ TEST(StarletARMCore, JitCompilesDrainWriteBufferNatively) EXPECT_EQ(core.GetJitFallbackInstructionCount(), 1u); } +TEST(StarletARMCore, JitCompilesHotCP15MaintenanceNatively) +{ +#if defined(_M_X86_64) + TestBus interpreter_bus; + TestBus jit_bus; + ARMCore interpreter(interpreter_bus); + ARMCore jit(jit_bus); + jit.SetJitEnabled(true); + const auto install_program = [](TestBus& bus) { + bus.WriteARM(0x00, 0xee033f10); // mcr p15, 0, r3, c3, c0, 0 (DACR) + bus.WriteARM(0x04, 0xee070f36); // mcr p15, 0, r0, c7, c6, 1 + bus.WriteARM(0x08, 0xee070f3a); // mcr p15, 0, r0, c7, c10, 1 + bus.WriteARM(0x0c, 0xeafffffe); // b . + }; + install_program(interpreter_bus); + install_program(jit_bus); + interpreter.SetRegister(3, 0x55555555); + jit.SetRegister(3, 0x55555555); + + ASSERT_EQ(interpreter.RunCycles(4), 4u); + ASSERT_EQ(jit.RunCycles(4), 4u); + EXPECT_EQ(jit.GetCP15State().domain_access_control, + interpreter.GetCP15State().domain_access_control); + EXPECT_EQ(jit.GetRegister(15), interpreter.GetRegister(15)); + EXPECT_EQ(jit.GetJitFallbackInstructionCount(), 0u); + EXPECT_EQ(jit.GetJitNativeExecutedInstructions(), 4u); +#endif +} + TEST(StarletARMCore, JitDefersCP15CacheInvalidationUntilTheHostBlockReturns) { TestBus bus; @@ -1041,18 +1252,324 @@ TEST(StarletARMCore, JitPreservedBlocksUseCurrentTLBGenerationForFastmem) EXPECT_EQ(core.RunCycles(2), 2u); ASSERT_EQ(core.GetRegister(1), 0x11223344u); - ASSERT_EQ(core.GetJitFallbackInstructionCount(), 1u); + ASSERT_EQ(core.GetJitFallbackInstructionCount(), 0u); ASSERT_EQ(core.GetJitCompiledBlockCount(), 1u); core.SetRegister(1, 0); core.SetRegister(15, 0x80000000); EXPECT_EQ(core.RunCycles(1), 1u); EXPECT_EQ(core.GetRegister(1), 0x11223344u); - EXPECT_EQ(core.GetJitFallbackInstructionCount(), 1u); + EXPECT_EQ(core.GetJitFallbackInstructionCount(), 0u); EXPECT_EQ(core.GetJitCompiledBlockCount(), 1u); #endif } +TEST(StarletARMCore, JitTLBRevalidationIsSharedByNativeBlocksOnTheSamePage) +{ +#if defined(_M_X86_64) + TestBus bus(0x10000); + ARMCore core(bus); + bus.WriteARM(0x0000, 0xe3a01001); // mov r1, #1 + bus.WriteARM(0x0020, 0xe3a02002); // mov r2, #2 + bus.WriteARM(0x0040, 0xee080f17); // invalidate unified TLB + bus.WriteARM(0x6000, 0x00000c02); // VA 0x80000000 section -> PA 0 + core.GetCP15State().translation_table_base = 0x4000; + core.GetCP15State().domain_access_control = 3; + core.GetCP15State().control |= 1; + core.SetJitEnabled(true); + + core.SetRegister(15, 0x80000000); + EXPECT_EQ(core.RunCycles(1), 1u); + core.SetRegister(15, 0x80000020); + EXPECT_EQ(core.RunCycles(1), 1u); + core.SetRegister(15, 0x80000040); + EXPECT_EQ(core.RunCycles(1), 1u); + + const u64 dispatches_before_revalidation = core.GetJitDispatchSlowCount(); + core.SetRegister(15, 0x80000000); + EXPECT_EQ(core.RunCycles(1), 1u); + EXPECT_EQ(core.GetJitDispatchSlowCount(), dispatches_before_revalidation); + + // The first block revalidated the unchanged page-table descriptor directly in generated code. + // A second native block on that physical page must likewise avoid a page-table walk and C++ + // block-map lookup. + core.SetRegister(15, 0x80000020); + EXPECT_EQ(core.RunCycles(1), 1u); + EXPECT_EQ(core.GetJitDispatchSlowCount(), dispatches_before_revalidation); +#endif +} + +TEST(StarletARMCore, JitSharedTLBRefillRejectsRemappedPhysicalCode) +{ +#if defined(_M_X86_64) + TestBus bus(0x110000); + ARMCore core(bus); + bus.WriteARM(0x000000, 0xe3a01001); // old page: mov r1, #1 + bus.WriteARM(0x000020, 0xe3a02002); // old page: mov r2, #2 + bus.WriteARM(0x000040, 0xee080f17); // invalidate unified TLB + bus.WriteARM(0x100000, 0xe3a01003); // new page: mov r1, #3 + bus.WriteARM(0x100020, 0xe3a02004); // new page: mov r2, #4 + bus.WriteARM(0x006000, 0x00000c02); // VA 0x80000000 section -> PA 0 + core.GetCP15State().translation_table_base = 0x4000; + core.GetCP15State().domain_access_control = 3; + core.GetCP15State().control |= 1; + core.SetJitEnabled(true); + + core.SetRegister(15, 0x80000000); + EXPECT_EQ(core.RunCycles(1), 1u); + core.SetRegister(15, 0x80000020); + EXPECT_EQ(core.RunCycles(1), 1u); + const size_t old_block_count = core.GetJitCompiledBlockCount(); + + // Change the page table under the still-valid TLB, then execute the architectural invalidation + // through the old mapping. The following dispatch must discover the new physical page. + bus.WriteARM(0x006000, 0x00100c02); + core.SetRegister(15, 0x80000040); + EXPECT_EQ(core.RunCycles(1), 1u); + core.SetRegister(15, 0x80000000); + EXPECT_EQ(core.RunCycles(1), 1u); + EXPECT_EQ(core.GetRegister(1), 3u); + core.SetRegister(15, 0x80000020); + EXPECT_EQ(core.RunCycles(1), 1u); + EXPECT_EQ(core.GetRegister(2), 4u); + EXPECT_EQ(core.GetJitCompiledBlockCount(), old_block_count + 3); +#endif +} + +TEST(StarletARMCore, JitRetainsPhysicalAliasesAcrossAddressSpaceSwitches) +{ +#if defined(_M_X86_64) + TestBus bus(0x110000); + ARMCore core(bus); + bus.WriteARM(0x000000, 0xe3a01001); // physical mapping 0: mov r1, #1 + bus.WriteARM(0x000020, 0xe3a02002); // physical mapping 0: mov r2, #2 + bus.WriteARM(0x000040, 0xee080f17); // invalidate unified TLB + bus.WriteARM(0x100000, 0xe3a01003); // physical mapping 1: mov r1, #3 + bus.WriteARM(0x100020, 0xe3a02004); // physical mapping 1: mov r2, #4 + bus.WriteARM(0x100040, 0xee080f17); // invalidate unified TLB + bus.WriteARM(0x006000, 0x00000c02); // VA 0x80000000 section -> PA 0 + core.GetCP15State().translation_table_base = 0x4000; + core.GetCP15State().domain_access_control = 3; + core.GetCP15State().control |= 1; + core.SetJitEnabled(true); + + for (const u32 address : {0x80000000U, 0x80000020U}) + { + core.SetRegister(15, address); + EXPECT_EQ(core.RunCycles(1), 1u); + } + + bus.WriteARM(0x006000, 0x00100c02); // Same virtual section -> PA 1 MiB. + core.SetRegister(15, 0x80000040); + EXPECT_EQ(core.RunCycles(1), 1u); + for (const u32 address : {0x80000000U, 0x80000020U}) + { + core.SetRegister(15, address); + EXPECT_EQ(core.RunCycles(1), 1u); + } + + // Return to the first address space. The first block refills the shared page translation; the + // second must immediately find its retained (MVA, physical page) entry instead of overwriting a + // single virtual-key slot and falling back to C++ again. + bus.WriteARM(0x006000, 0x00000c02); + core.SetRegister(15, 0x80000040); + EXPECT_EQ(core.RunCycles(1), 1u); + const u64 dispatches_before_refill = core.GetJitDispatchSlowCount(); + core.SetRegister(15, 0x80000000); + EXPECT_EQ(core.RunCycles(1), 1u); + EXPECT_EQ(core.GetRegister(1), 1u); + core.SetRegister(15, 0x80000020); + EXPECT_EQ(core.RunCycles(1), 1u); + EXPECT_EQ(core.GetRegister(2), 2u); + EXPECT_EQ(core.GetJitDispatchSlowCount(), dispatches_before_refill + 1); +#endif +} + +TEST(StarletARMCore, JitFastBlockCacheRetainsFourCollidingHotBlocks) +{ +#if defined(_M_X86_64) + TestBus bus(0xd0000); + ARMCore core(bus); + core.SetJitEnabled(true); + + // These ARM addresses deliberately have the same upper 16 bits after multiplying by the JIT + // cache's 0x9e3779b1 hash constant. They therefore occupy the four ways of one cache set. + constexpr std::array addresses = {0x000014, 0x04cb94, 0x07e168, 0x099714, 0x0cace8}; + for (const u32 address : addresses) + bus.WriteARM(address, 0xe3a01001); // mov r1, #1 + + for (size_t i = 0; i < 4; ++i) + { + const u32 address = addresses[i]; + core.SetRegister(15, address); + EXPECT_EQ(core.RunCycles(1), 1u); + } + + const u64 dispatches_after_fill = core.GetJitDispatchSlowCount(); + const u64 collisions_after_fill = core.GetJitDispatchCollisionCount(); + for (size_t i = 0; i < 4; ++i) + { + const u32 address = addresses[i]; + core.SetRegister(15, address); + EXPECT_EQ(core.RunCycles(1), 1u); + } + EXPECT_EQ(core.GetJitDispatchSlowCount(), dispatches_after_fill); + EXPECT_EQ(core.GetJitDispatchCollisionCount(), collisions_after_fill); + + // A fifth distinct key proves that the set is actually full and exercises bounded replacement. + core.SetRegister(15, addresses.back()); + EXPECT_EQ(core.RunCycles(1), 1u); + EXPECT_EQ(core.GetJitDispatchSlowCount(), dispatches_after_fill + 1); + EXPECT_EQ(core.GetJitDispatchCollisionCount(), collisions_after_fill + 1); +#endif +} + +TEST(StarletARMCore, JitCachesFallbackOnlyBlocks) +{ +#if defined(_M_X86_64) + TestBus bus(0x10000); + ARMCore core(bus); + // MUL uses the exact interpreter helper in this JIT. A block beginning with it therefore has + // zero directly emitted ARM instructions, but its generated fallback wrapper is still reusable. + bus.WriteARM(0x0000, 0xe0010190); // mul r1, r0, r1 + core.SetJitEnabled(true); + + core.SetRegister(0, 3); + core.SetRegister(1, 4); + core.SetRegister(15, 0); + EXPECT_EQ(core.RunCycles(1), 1u); + EXPECT_EQ(core.GetRegister(1), 12u); + const u64 dispatches_after_compile = core.GetJitDispatchSlowCount(); + const u64 fallbacks_after_compile = core.GetJitFallbackInstructionCount(); + + core.SetRegister(1, 5); + core.SetRegister(15, 0); + EXPECT_EQ(core.RunCycles(1), 1u); + EXPECT_EQ(core.GetRegister(1), 15u); + EXPECT_EQ(core.GetJitFallbackInstructionCount(), fallbacks_after_compile + 1); + EXPECT_EQ(core.GetJitDispatchSlowCount(), dispatches_after_compile); +#endif +} + +TEST(StarletARMCore, ARMJitCompilesLogicalImmediateAndShiftCarry) +{ +#if defined(_M_X86_64) + TestBus interpreter_bus(0x1000); + TestBus jit_bus(0x1000); + ARMCore interpreter(interpreter_bus); + ARMCore jit(jit_bus); + jit.SetJitEnabled(true); + const auto install_program = [](TestBus& bus) { + bus.WriteARM(0x00, 0xe3180701); // tst r8, #0x40000; rotated immediate supplies C + bus.WriteARM(0x04, 0xeafffffe); // b . + bus.WriteARM(0x20, 0xe1b02820); // movs r2, r0, lsr #16; bit 15 supplies C + bus.WriteARM(0x24, 0xeafffffe); // b . + }; + install_program(interpreter_bus); + install_program(jit_bus); + + for (ARMCore* core : {&interpreter, &jit}) + { + core->SetCPSR(static_cast(ARMCore::Mode::System) | ARMCore::CPSR_C | ARMCore::CPSR_V); + core->SetRegister(8, 0x40000); + core->SetRegister(15, 0); + } + ASSERT_EQ(interpreter.RunCycles(2), 2u); + ASSERT_EQ(jit.RunCycles(2), 2u); + EXPECT_EQ(jit.GetCPSR(), interpreter.GetCPSR()); + + for (ARMCore* core : {&interpreter, &jit}) + { + core->SetCPSR(static_cast(ARMCore::Mode::System) | ARMCore::CPSR_V); + core->SetRegister(0, 0x80018000); + core->SetRegister(2, 0); + core->SetRegister(15, 0x20); + } + ASSERT_EQ(interpreter.RunCycles(2), 2u); + ASSERT_EQ(jit.RunCycles(2), 2u); + EXPECT_EQ(jit.GetRegister(2), interpreter.GetRegister(2)); + EXPECT_EQ(jit.GetCPSR(), interpreter.GetCPSR()); + EXPECT_EQ(jit.GetJitFallbackInstructionCount(), 0u); + EXPECT_EQ(jit.GetJitNativeExecutedInstructions(), 4u); +#endif +} + +TEST(StarletARMCore, ARMJitCompilesIRQVectorLoadPCWithInterworking) +{ +#if defined(_M_X86_64) + TestBus bus; + ARMCore core(bus); + bus.SetFastmemEnabled(true); + bus.WriteARM(0x00, 0xe59ff018); // ldr pc, [pc, #0x18] -> 0x20 + bus.WriteARM(0x20, 0x00000101); // enter Thumb at 0x100 + core.SetRegister(15, 0); + core.SetJitEnabled(true); + + EXPECT_EQ(core.RunCycles(1), 1u); + EXPECT_EQ(core.GetRegister(15), 0x100u); + EXPECT_NE(core.GetCPSR() & ARMCore::CPSR_T, 0u); + EXPECT_EQ(core.GetJitFallbackInstructionCount(), 0u); + EXPECT_EQ(core.GetJitSlowReadCount(), 0u); + EXPECT_EQ(core.GetJitNativeExecutedInstructions(), 1u); + EXPECT_TRUE(bus.SRAMCanariesIntact()); +#endif +} + +TEST(StarletARMCore, ARMJitCompilesLongMultiplyFamily) +{ +#if defined(_M_X86_64) + struct Case + { + u32 instruction; + u32 cpsr; + u32 rm; + u32 rs; + u32 rd_hi; + u32 rd_lo; + }; + constexpr std::array cases = { + Case{0xe0834291, static_cast(ARMCore::Mode::System) | ARMCore::CPSR_C, 0x10000, 0x10001, + 0, 0}, // UMULL + Case{0xe0c34291, static_cast(ARMCore::Mode::System) | ARMCore::CPSR_V, 0xfffffff0, 0x10, + 0, 0}, // SMULL + Case{0xe0b34291, static_cast(ARMCore::Mode::System) | ARMCore::CPSR_C | ARMCore::CPSR_V, + 0xffffffff, 1, 0, 1}, // UMLALS, result wraps to zero and preserves CV + Case{0x10834291, static_cast(ARMCore::Mode::System) | ARMCore::CPSR_Z, 7, 9, 0x11223344, + 0x55667788}, // UMULLNE, predicate fails + }; + + for (const Case& test : cases) + { + TestBus interpreter_bus; + TestBus jit_bus; + ARMCore interpreter(interpreter_bus); + ARMCore jit(jit_bus); + interpreter_bus.WriteARM(0, test.instruction); + interpreter_bus.WriteARM(4, 0xeafffffe); // b . + jit_bus.WriteARM(0, test.instruction); + jit_bus.WriteARM(4, 0xeafffffe); // b . + for (ARMCore* core : {&interpreter, &jit}) + { + core->SetCPSR(test.cpsr); + core->SetRegister(1, test.rm); + core->SetRegister(2, test.rs); + core->SetRegister(3, test.rd_hi); + core->SetRegister(4, test.rd_lo); + core->SetRegister(15, 0); + } + jit.SetJitEnabled(true); + + ASSERT_EQ(interpreter.RunCycles(2), 2u); + ASSERT_EQ(jit.RunCycles(2), 2u); + EXPECT_EQ(jit.GetRegister(3), interpreter.GetRegister(3)); + EXPECT_EQ(jit.GetRegister(4), interpreter.GetRegister(4)); + EXPECT_EQ(jit.GetCPSR(), interpreter.GetCPSR()); + EXPECT_EQ(jit.GetJitFallbackInstructionCount(), 0u); + EXPECT_EQ(jit.GetJitNativeExecutedInstructions(), 2u); + } +#endif +} + TEST(StarletARMCore, FCSESwitchPreservesTaggedTLBTranslations) { TestBus bus(0x10000); diff --git a/Source/UnitTests/Core/PowerPC/Jit64Common/DCache.cpp b/Source/UnitTests/Core/PowerPC/Jit64Common/DCache.cpp new file mode 100644 index 0000000000..a2827049c0 --- /dev/null +++ b/Source/UnitTests/Core/PowerPC/Jit64Common/DCache.cpp @@ -0,0 +1,316 @@ +// Copyright 2026 Dolphin Emulator Project +// SPDX-License-Identifier: GPL-2.0-or-later + +#include +#include + +#include "Common/x64ABI.h" +#include "Core/ConfigManager.h" +#include "Core/Core.h" +#include "Core/HW/Memmap.h" +#include "Core/PowerPC/Jit64/Jit.h" +#include "Core/PowerPC/Jit64Common/Jit64AsmCommon.h" +#include "Core/PowerPC/Jit64Common/Jit64Constants.h" +#include "Core/PowerPC/MMU.h" +#include "Core/PowerPC/PowerPC.h" +#include "Core/System.h" + +#include + +namespace +{ +using namespace Gen; + +// Execute the real SafeLoad/SafeWrite emitter, including its fallback and register contract. +// Testing only the C++ MMU helpers cannot detect corruption introduced at the native call site. +class DCacheCode : public CommonAsmRoutines +{ +public: + explicit DCacheCode(Core::System& system) : CommonAsmRoutines(jit), jit(system) + { + jit.jo = {}; + jit.js = {}; + AllocCodeSpace(512 * 1024); + old_read = dcache32_read_hit_dbat; + old_write = dcache32_write_hit_dbat; + dcache32_read_hit_dbat = AlignCode4(); + GenDCache32Hit(false); + dcache32_write_hit_dbat = AlignCode4(); + GenDCache32Hit(true); + } + + ~DCacheCode() override + { + dcache32_read_hit_dbat = old_read; + dcache32_write_hit_dbat = old_write; + } + + void Access(bool write, u32 address, u32 value, X64Reg address_reg, X64Reg value_reg, + s32 offset = 0, int flags = 0) + { + run = reinterpret_cast(AlignCode4()); + ABI_PushRegistersAndAdjustStack(ABI_ALL_CALLEE_SAVED, 8); + MOV(64, R(RPPCSTATE), ImmPtr(reinterpret_cast(&jit.m_ppc_state) + 0x80)); + MOV(32, R(RDX), Imm32(0x12345678)); + MOV(32, R(RCX), Imm32(0x87654321)); + MOV(32, R(address_reg), Imm32(address)); + if (write) + MOV(32, R(value_reg), Imm32(value)); + const BitSet32 live{RCX, RDX, R8, R9}; + if (flags & SAFE_LOADSTORE_NO_PROLOG) + SUB(64, R(RSP), Imm8(8)); + if (write) + SafeWriteRegToReg(value_reg, address_reg, 32, offset, live, flags); + else + SafeLoadToReg(value_reg, R(address_reg), 32, offset, live, false, flags); + if (flags & SAFE_LOADSTORE_NO_PROLOG) + ADD(64, R(RSP), Imm8(8)); + + MOV(64, R(R11), ImmPtr(result.data())); + for (const auto reg : {RAX, RCX, RDX, R8, R9}) + MOV(64, MDisp(R11, static_cast(reg) * sizeof(u64)), R(reg)); + ABI_PopRegistersAndAdjustStack(ABI_ALL_CALLEE_SAVED, 8); + RET(); + ASSERT_FALSE(HasWriteFailed()); + run(); + } + + std::array result{}; + void (*run)() = nullptr; + Jit64 jit; + const u8* old_read; + const u8* old_write; +}; + +class Jit64DCache : public testing::Test +{ +protected: + static void SetUpTestSuite() + { + SConfig::Init(); + auto& system = Core::System::GetInstance(); + system.SetIsWii(true); + system.GetMemory().Init(); + Core::DeclareAsCPUThread(); + system.GetPPCState().dCache.Init(system.GetMemory()); + } + + static void TearDownTestSuite() + { + auto& system = Core::System::GetInstance(); + system.GetPPCState().m_enable_dcache = false; + system.GetMemory().Shutdown(); + system.SetIsWii(false); + Core::UndeclareAsCPUThread(); + SConfig::Shutdown(); + } + + void SetUp() override + { + state.dCache.Reset(); + state.m_enable_dcache = true; + state.msr.DR = 1; + state.feature_flags = FEATURE_FLAG_MSR_DR; + state.Exceptions = 0; + state.spr[SPR_HID0] = 0; + bats.fill(0); + bats[0x80000000 >> PowerPC::BAT_INDEX_SHIFT] = PowerPC::BAT_MAPPED_BIT; + bats[0x90000000 >> PowerPC::BAT_INDEX_SHIFT] = 0x10000000 | PowerPC::BAT_MAPPED_BIT; + code = std::make_unique(system); + } + + void TearDown() override { code.reset(); } + + Core::System& system = Core::System::GetInstance(); + PowerPC::PowerPCState& state = system.GetPPCState(); + PowerPC::MMU& mmu = system.GetMMU(); + Memory::MemoryManager& memory = system.GetMemory(); + PowerPC::BatTable& bats = const_cast(mmu.GetDBATTable()); + std::unique_ptr code; +}; + +TEST_F(Jit64DCache, StoreMissRetainsAddressInRDX) +{ + // The data is deliberately also a mapped address: a broken fallback writes there instead. + memory.Write_U32(0, 0x1000); + memory.Write_U32(0, 0x4000); + code->Access(true, 0x80001000, 0x80004000, RDX, RCX); + EXPECT_EQ(mmu.ReadForJit(0x80001000), 0x80004000); + EXPECT_EQ(mmu.ReadForJit(0x80004000), 0U); + EXPECT_EQ(code->result[RDX], 0x80001000); + EXPECT_EQ(code->result[RCX], 0x80004000); +} + +TEST_F(Jit64DCache, LoadMissRetainsAddressInRDX) +{ + memory.Write_U32(0x89abcdef, 0x1000); + code->Access(false, 0x80001000, 0, RDX, R8); + EXPECT_EQ(code->result[R8], 0x89abcdef); + EXPECT_EQ(code->result[RDX], 0x80001000); +} + +TEST_F(Jit64DCache, HitPreservesLiveScratchRegisters) +{ + mmu.WriteForJit(0xffffffff, 0x80001000); + code->Access(false, 0x80001000, 0, R9, R8); + EXPECT_EQ(code->result[R8], 0xffffffff); + EXPECT_EQ(code->result[RDX], 0x12345678U); + EXPECT_EQ(code->result[RCX], 0x87654321U); + code->Access(true, 0x80001000, 0x11223344, R9, R8); + EXPECT_EQ(mmu.ReadForJit(0x80001000), 0x11223344U); + EXPECT_EQ(code->result[RDX], 0x12345678U); + EXPECT_EQ(code->result[RCX], 0x87654321U); +} + +TEST_F(Jit64DCache, PhysicalAccessIgnoresDBAT) +{ + state.msr.DR = 0; + state.feature_flags = CPUEmuFeatureFlags{}; + bats[0] = 0x20000 | PowerPC::BAT_MAPPED_BIT; + mmu.WriteForJit(0x11111111, 0x1000); + mmu.WriteForJit(0x22222222, 0x21000); + code->Access(false, 0x1000, 0, R9, R8); + EXPECT_EQ(code->result[R8], 0x11111111U); + code->Access(true, 0x1000, 0x33333333, R9, R8); + EXPECT_EQ(mmu.ReadForJit(0x1000), 0x33333333U); + EXPECT_EQ(mmu.ReadForJit(0x21000), 0x22222222U); +} + +TEST_F(Jit64DCache, LoadRegisterAndOffsetCombinations) +{ + for (const bool hit : {false, true}) + { + for (const auto address_reg : {RAX, RCX, RDX, R8, R9}) + { + for (const auto value_reg : {RAX, RCX, RDX, R8, R9}) + { + for (const s32 offset : {0, 4, -4}) + { + SCOPED_TRACE(testing::Message() + << hit << ' ' << address_reg << ' ' << value_reg << ' ' << offset); + state.dCache.Reset(); + memory.Write_U32(0xfedcba98, 0x1020); + if (hit) + ASSERT_EQ(mmu.ReadForJit(0x80001020), 0xfedcba98); + code->Access(false, 0x80001020 - offset, 0, address_reg, value_reg, offset); + EXPECT_EQ(code->result[value_reg], 0xfedcba98); + } + } + } + } +} + +TEST_F(Jit64DCache, StoreRegisterAndOffsetCombinations) +{ + for (const bool hit : {false, true}) + { + for (const auto address_reg : {RAX, RCX, RDX, R8, R9}) + { + for (const auto value_reg : {RCX, RDX, R8, R9}) + { + if (address_reg == value_reg) + continue; + for (const s32 offset : {0, 4, -4}) + { + SCOPED_TRACE(testing::Message() + << hit << ' ' << address_reg << ' ' << value_reg << ' ' << offset); + state.dCache.Reset(); + memory.Write_U32(0, 0x1020); + if (hit) + ASSERT_EQ(mmu.ReadForJit(0x80001020), 0U); + code->Access(true, 0x80001020 - offset, 0xfedcba98, address_reg, value_reg, offset, + EmuCodeBlock::SAFE_LOADSTORE_CLOBBER_RSCRATCH_INSTEAD_OF_ADDR); + EXPECT_EQ(mmu.ReadForJit(0x80001020), 0xfedcba98); + EXPECT_EQ(code->result[value_reg], 0xfedcba98); + } + } + } + } +} + +TEST_F(Jit64DCache, SplitLineAndPageAccess) +{ + for (const u32 offset : {29U, 30U, 31U, 4093U, 4094U, 4095U}) + { + SCOPED_TRACE(offset); + const u32 address = 0x80001000 + offset; + mmu.WriteForJit(0x11223344, address); + code->Access(false, address, 0, RDX, R8); + EXPECT_EQ(code->result[R8], 0x11223344U); + code->Access(true, address, 0x55667788, RDX, RCX); + EXPECT_EQ(mmu.ReadForJit(address), 0x55667788U); + } +} + +TEST_F(Jit64DCache, InhibitedBATBypassesCachedData) +{ + mmu.WriteForJit(0x11111111, 0x80001000); + memory.Write_U32(0x22222222, 0x1000); + bats[0x80000000 >> PowerPC::BAT_INDEX_SHIFT] |= PowerPC::BAT_WI_BIT; + code->Access(false, 0x80001000, 0, RDX, R8); + EXPECT_EQ(code->result[R8], 0x22222222U); + code->Access(true, 0x80001000, 0x33333333, RDX, RCX); + EXPECT_EQ(memory.Read_U32(0x1000), 0x33333333U); + bats[0x80000000 >> PowerPC::BAT_INDEX_SHIFT] &= ~PowerPC::BAT_WI_BIT; + EXPECT_EQ(mmu.ReadForJit(0x80001000), 0x11111111U); +} + +TEST_F(Jit64DCache, SharedRoutineStackAndDRFlag) +{ + // Shared paired-load routines are emitted without a known block's feature flags and run + // with a return address on the stack. DR_ON is their explicit translation contract. + state.feature_flags = CPUEmuFeatureFlags{}; + constexpr int flags = EmuCodeBlock::SAFE_LOADSTORE_NO_PROLOG | + EmuCodeBlock::SAFE_LOADSTORE_NO_UPDATE_PC | + EmuCodeBlock::SAFE_LOADSTORE_DR_ON; + memory.Write_U32(0x89abcdef, 0x1000); + for (int repeat = 0; repeat < 2; ++repeat) + { + code->Access(false, 0x80001000, 0, RDX, R8, 0, flags); + EXPECT_EQ(code->result[R8], 0x89abcdef); + code->Access(true, 0x80001000, 0x89abcdef, RDX, RCX, 0, flags); + EXPECT_EQ(mmu.ReadForJit(0x80001000), 0x89abcdef); + } +} + +TEST_F(Jit64DCache, NativePLRUAndDirtyStateMatchMMU) +{ + for (const u32 base : {0x80000000U, 0x90000000U}) + { + state.dCache.Reset(); + constexpr u32 set = 126; + for (u32 way = 0; way < PowerPC::CACHE_WAYS; ++way) + mmu.WriteForJit(0xffffffff, base + way * 4096 + set * 32); + for (u32 way = 0; way < PowerPC::CACHE_WAYS; ++way) + { + const u32 address = base + way * 4096 + set * 32; + for (const bool write : {false, true}) + { + code->Access(write, address, 0xffffffff, RDX, R8); + for (u32 old_plru = 0; old_plru < 128; ++old_plru) + { + SCOPED_TRACE(testing::Message() << base << ' ' << way << ' ' << write << ' ' << old_plru); + auto& cache = state.dCache; + cache.plru[set] = static_cast(old_plru); + cache.modified[set] = 0x80; + if (write) + ASSERT_TRUE(mmu.TryWriteDCache32ForJit(0xffffffff, address)); + else + ASSERT_EQ(mmu.TryReadDCache32ForJit(address), 0x100000000ULL); + const auto expected_plru = cache.plru[set]; + const auto expected_dirty = cache.modified[set]; + const auto expected_data = cache.data[set]; + cache.plru[set] = static_cast(old_plru); + cache.modified[set] = 0x80; + code->run(); + EXPECT_EQ(cache.plru[set], expected_plru); + EXPECT_EQ(cache.modified[set], expected_dirty); + EXPECT_EQ(cache.data[set], expected_data); + if (!write) + EXPECT_EQ(code->result[R8], 0xffffffff); + } + } + } + } +} +} // namespace