Sdl backend (#670)

* [audio] added sdl audio backend and in-tree atrac9 decoder

* [input] replaced per-platform pad readers with sdl gamepad input

* [video] added sdl window and host display plumbing

* [gui] added host display options and per-game render settings

* [bink] synced host movie playback to the guest audio clock

* [cpu] hooked windows write faults into guest image tracking

* [perf] added guest and render profiling, reserved host cpu lanes

* [kernel] fixed stale pthread mutex handle alias

* [host] wired the sdl session, save-data paths and project references

* [audio] hoisted ajm trace stackalloc out of its loop

* [video] Add guest image sync setting

* [build] Strip native symbols

* reuse
This commit is contained in:
Berk
2026-07-28 03:33:26 +03:00
committed by GitHub
parent b4cc5f88ca
commit 2b6bd5a532
111 changed files with 9846 additions and 4479 deletions
@@ -23,14 +23,71 @@ public sealed partial class DirectExecutionBackend
private static long _perfHleTotal;
private static long _perfHleDispatchTicks;
private sealed class PerfHleExportCost
{
public long Calls;
public long Ticks;
}
private static readonly System.Collections.Concurrent.ConcurrentDictionary<string, PerfHleExportCost> _perfHleCosts = new();
/// <summary>
/// Name of the export currently being dispatched on this thread, so the
/// gateway can attribute its elapsed time once the call returns. Answering
/// "which export is worth optimising" needs cost per export, not just call
/// counts — a rare expensive call and a hot cheap one look identical in a
/// frequency histogram.
/// </summary>
[System.ThreadStatic]
private static string? _perfHleCurrentExport;
private static long _perfHleFirstTimestamp;
private static void RecordPerfHleDispatchTime(long ticks)
{
var total = System.Threading.Interlocked.Add(ref _perfHleDispatchTicks, ticks);
var calls = System.Threading.Interlocked.Read(ref _perfHleTotal);
var name = _perfHleCurrentExport;
if (name is not null)
{
var cost = _perfHleCosts.GetOrAdd(name, static _ => new PerfHleExportCost());
System.Threading.Interlocked.Increment(ref cost.Calls);
System.Threading.Interlocked.Add(ref cost.Ticks, ticks);
}
if (calls > 0 && calls % 500000 == 0)
{
var avgUs = (double)total / System.Diagnostics.Stopwatch.Frequency * 1_000_000.0 / calls;
System.Console.Error.WriteLine($"[PERF][HLE] managed_dispatch_avg={avgUs:F3}us total_managed_s={(double)total / System.Diagnostics.Stopwatch.Frequency:F2}");
var frequency = (double)System.Diagnostics.Stopwatch.Frequency;
var avgUs = (double)total / frequency * 1_000_000.0 / calls;
var first = System.Threading.Interlocked.CompareExchange(ref _perfHleFirstTimestamp, 0, 0);
var wallSeconds = first == 0
? 0
: (double)(System.Diagnostics.Stopwatch.GetTimestamp() - first) / frequency;
System.Console.Error.WriteLine(
$"[PERF][HLE] managed_dispatch_avg={avgUs:F3}us " +
$"total_managed_s={(double)total / frequency:F2} " +
$"wall_s={wallSeconds:F2} " +
$"cores={(wallSeconds > 0 ? total / frequency / wallSeconds : 0):F2}");
var snapshot = new System.Collections.Generic.List<System.Collections.Generic.KeyValuePair<string, PerfHleExportCost>>(_perfHleCosts.Count + 16);
foreach (var kvp in _perfHleCosts)
{
snapshot.Add(kvp);
}
var top = snapshot
.OrderByDescending(kvp => System.Threading.Interlocked.Read(ref kvp.Value.Ticks))
.Take(12)
.Select(kvp =>
{
var seconds = System.Threading.Interlocked.Read(ref kvp.Value.Ticks) / frequency;
var callCount = System.Threading.Interlocked.Read(ref kvp.Value.Calls);
var cores = wallSeconds > 0 ? seconds / wallSeconds : 0;
var perCallUs = callCount > 0 ? seconds * 1_000_000.0 / callCount : 0;
return $"{kvp.Key}: {cores:F2}cores {seconds:F1}s n={callCount} {perCallUs:F2}us/call";
});
System.Console.Error.WriteLine($"[PERF][HLE] cost: {string.Join(" | ", top)}");
}
}
@@ -39,7 +96,16 @@ public sealed partial class DirectExecutionBackend
private static void RecordPerfHleCall(string name)
{
_perfHleCurrentExport = name;
var total = System.Threading.Interlocked.Increment(ref _perfHleTotal);
if (total == 1)
{
System.Threading.Interlocked.CompareExchange(
ref _perfHleFirstTimestamp,
System.Diagnostics.Stopwatch.GetTimestamp(),
0);
}
if (!_perfHleNoDict)
{
_perfHleCounts.AddOrUpdate(name, 1, static (_, v) => v + 1);
@@ -40,6 +40,15 @@ public sealed partial class DirectExecutionBackend
}
_rawExceptionHandler = (nint)AddVectoredExceptionHandler(1u, _rawExceptionHandlerStub);
Console.Error.WriteLine($"[LOADER][INFO] Raw exception handler installed: 0x{_rawExceptionHandler:X16}");
// The raw handler carries the guest-image write-fault bridge, so the
// path must be compiled before the first protected-page store can
// reach it. Guest code has not started yet, so warming here cannot
// race a real fault.
SharpEmu.HLE.GuestImageWriteTracker.WarmUp();
Console.Error.WriteLine(
"[LOADER][INFO] Guest image CPU write tracking: " +
$"{(SharpEmu.HLE.GuestImageWriteTracker.Enabled ? "enabled" : "disabled")}");
}
else
{
@@ -0,0 +1,322 @@
// Copyright (C) 2026 SharpEmu Emulator Project
// SPDX-License-Identifier: GPL-2.0-or-later
using System;
using System.Collections.Concurrent;
using System.Collections.Generic;
using System.Diagnostics;
using System.Linq;
using System.Threading;
namespace SharpEmu.Core.Cpu.Native;
/// <summary>
/// Sampling profiler for guest code. Managed profilers only see the emulator's
/// own frames — once a guest thread is running translated code it is opaque to
/// them, so a title that burns its cores inside its own spin loops looks like
/// unattributed native time. This walks the guest thread registry and samples
/// each thread's host RIP, which lands directly on the guest instruction being
/// executed.
/// </summary>
public sealed partial class DirectExecutionBackend
{
private static readonly bool _profileGuestRip =
string.Equals(
Environment.GetEnvironmentVariable("SHARPEMU_PROFILE_GUEST_RIP"),
"1",
StringComparison.Ordinal);
private static readonly int _profileGuestRipIntervalMs =
int.TryParse(
Environment.GetEnvironmentVariable("SHARPEMU_PROFILE_GUEST_RIP_INTERVAL_MS"),
out var interval) && interval > 0
? interval
: 2;
private static readonly int _profileGuestRipReportSeconds =
int.TryParse(
Environment.GetEnvironmentVariable("SHARPEMU_PROFILE_GUEST_RIP_REPORT_S"),
out var report) && report > 0
? report
: 15;
private const ulong GuestImageBase = 0x0000_0008_0000_0000UL;
private const ulong GuestImageLimit = 0x0000_0009_0000_0000UL;
private int _guestRipSamplerStarted;
private readonly ConcurrentDictionary<ulong, long> _guestRipSamples = new();
private readonly ConcurrentDictionary<string, long> _guestRipThreadSamples = new();
private readonly ConcurrentDictionary<string, long> _guestWaitSamples = new();
private readonly ConcurrentDictionary<string, long> _guestThreadWaitSamples = new();
private long _guestRipTotalSamples;
private long _guestWaitTotalSamples;
private long _guestRipCaptureFailures;
private long _guestRipSamplerErrors;
private int _guestRipSampleCursor;
/// <summary>
/// Names the HLE call a thread is parked in, using the guest RIP the import
/// dispatcher left on its context.
/// </summary>
private string ResolveWaitLabel(GuestThreadState thread)
{
var context = thread.Context;
if (context is null)
{
return "<no-context>";
}
var importIndex = context.ActiveImportIndex;
if ((uint)importIndex >= (uint)_importEntries.Length)
{
// Host code with no import in flight: the thread is parked by the
// emulator's own scheduler. The cooperative block records why, which
// is the part that actually identifies what the frame is waiting on.
var blockReason = thread.BlockReason;
return string.IsNullOrEmpty(blockReason)
? "<idle-or-scheduler>"
: $"blocked:{blockReason}";
}
var entry = _importEntries[importIndex];
return entry.Export?.Name ?? entry.Nid;
}
internal void ClearActiveImportIndex()
{
if (!_profileGuestRip)
{
return;
}
var context = ActiveCpuContext;
if (context is not null)
{
context.ActiveImportIndex = -1;
}
}
private void EnsureGuestRipSampler()
{
if (!_profileGuestRip ||
!OperatingSystem.IsWindows() ||
Interlocked.Exchange(ref _guestRipSamplerStarted, 1) != 0)
{
return;
}
var sampler = new Thread(GuestRipSampleLoop)
{
IsBackground = true,
Name = "SharpEmu guest RIP sampler",
// Sampling suspends guest threads briefly. Keep this diagnostic below
// the title workers so it observes them without becoming the bottleneck.
Priority = ThreadPriority.BelowNormal,
};
sampler.Start();
Console.Error.WriteLine(
$"[PERF][GUEST] RIP sampler started: interval={_profileGuestRipIntervalMs}ms " +
$"report={_profileGuestRipReportSeconds}s");
}
private void GuestRipSampleLoop()
{
var clock = Stopwatch.StartNew();
var lastReportMs = 0L;
var lastReportSamples = 0L;
while (true)
{
try
{
var guestThreads = SnapshotGuestThreads();
var sampleIndex = guestThreads.Length == 0
? 0
: (int)((uint)Interlocked.Increment(ref _guestRipSampleCursor) % (uint)guestThreads.Length);
foreach (var thread in guestThreads.Skip(sampleIndex).Take(1))
{
var hostThreadId = Volatile.Read(ref thread.HostThreadId);
if (hostThreadId == 0)
{
continue;
}
if (!TryCaptureHostThreadContext(hostThreadId, out var snapshot) ||
!snapshot.IsValid)
{
Interlocked.Increment(ref _guestRipCaptureFailures);
continue;
}
_guestRipSamples.AddOrUpdate(snapshot.Rip, 1, static (_, value) => value + 1);
_guestRipThreadSamples.AddOrUpdate(
string.IsNullOrEmpty(thread.Name) ? "<unnamed>" : thread.Name,
1,
static (_, value) => value + 1);
Interlocked.Increment(ref _guestRipTotalSamples);
// A host RIP means the thread is inside the emulator rather
// than running translated code. DispatchImport parks the
// guest RIP on the import stub for the call being serviced,
// so the stub address names what the thread is waiting on —
// no hot-path bookkeeping needed to find out.
if (snapshot.Rip >= GuestImageBase && snapshot.Rip < GuestImageLimit)
{
continue;
}
_guestWaitSamples.AddOrUpdate(
ResolveWaitLabel(thread),
1,
static (_, value) => value + 1);
_guestThreadWaitSamples.AddOrUpdate(
string.IsNullOrEmpty(thread.Name) ? "<unnamed>" : thread.Name,
1,
static (_, value) => value + 1);
Interlocked.Increment(ref _guestWaitTotalSamples);
}
Thread.Sleep(_profileGuestRipIntervalMs);
var elapsedMs = clock.ElapsedMilliseconds;
if (elapsedMs - lastReportMs < _profileGuestRipReportSeconds * 1000L)
{
continue;
}
var samples = Interlocked.Read(ref _guestRipTotalSamples);
ReportGuestRipSamples(samples - lastReportSamples, (elapsedMs - lastReportMs) / 1000.0);
lastReportMs = elapsedMs;
lastReportSamples = samples;
}
catch (Exception exception)
{
// A title can tear down a thread or its context during a capture.
// The profiler must never silently die or affect guest execution.
if (Interlocked.Increment(ref _guestRipSamplerErrors) == 1)
{
Console.Error.WriteLine($"[PERF][GUEST] sampler recovery: {exception.GetType().Name}: {exception.Message}");
}
}
}
}
private void ReportGuestRipSamples(long windowSamples, double windowSeconds)
{
var total = Interlocked.Read(ref _guestRipTotalSamples);
if (total == 0)
{
return;
}
var byRip = new List<KeyValuePair<ulong, long>>(_guestRipSamples.Count + 16);
foreach (var pair in _guestRipSamples)
{
byRip.Add(pair);
}
// A tight spin lands on a handful of instructions; grouping by 4 KB page
// as well shows which routine those instructions belong to.
var byPage = new Dictionary<ulong, long>();
foreach (var pair in byRip)
{
var page = pair.Key & ~0xFFFUL;
byPage[page] = byPage.TryGetValue(page, out var existing)
? existing + pair.Value
: pair.Value;
}
var byThread = new List<KeyValuePair<string, long>>(_guestRipThreadSamples.Count + 16);
foreach (var pair in _guestRipThreadSamples)
{
byThread.Add(pair);
}
Console.Error.WriteLine(
$"[PERF][GUEST] samples={total} window={windowSamples} in {windowSeconds:F1}s " +
$"capture_failures={Interlocked.Read(ref _guestRipCaptureFailures)}");
Console.Error.WriteLine(
"[PERF][GUEST] top_rip: " +
string.Join(
" | ",
byRip.OrderByDescending(pair => pair.Value)
.Take(12)
.Select(pair =>
$"0x{pair.Key:X}{DescribeGuestAddress(pair.Key)}={pair.Value * 100.0 / total:F1}%")));
Console.Error.WriteLine(
"[PERF][GUEST] top_page: " +
string.Join(
" | ",
byPage.OrderByDescending(pair => pair.Value)
.Take(8)
.Select(pair =>
$"0x{pair.Key:X}{DescribeGuestAddress(pair.Key)}={pair.Value * 100.0 / total:F1}%")));
var byWait = new List<KeyValuePair<string, long>>(_guestWaitSamples.Count + 16);
foreach (var pair in _guestWaitSamples)
{
byWait.Add(pair);
}
var waitTotal = Interlocked.Read(ref _guestWaitTotalSamples);
Console.Error.WriteLine(
$"[PERF][GUEST] waiting={waitTotal * 100.0 / total:F1}% of guest thread-time; top_wait: " +
string.Join(
" | ",
byWait.OrderByDescending(pair => pair.Value)
.Take(12)
.Select(pair => $"{pair.Key}={pair.Value * 100.0 / total:F1}%")));
// Per-thread spin/park split. The global wait share mixes the job pool in
// with a dozen dormant threads, which hides the number that matters:
// how much of a core each worker actually burns.
Console.Error.WriteLine(
"[PERF][GUEST] thread_split (running/parked): " +
string.Join(
" | ",
byThread.OrderByDescending(pair => pair.Value)
.Take(10)
.Select(pair =>
{
var parked = _guestThreadWaitSamples.TryGetValue(pair.Key, out var wait) ? wait : 0;
var running = pair.Value - parked;
return $"{pair.Key}={running * 100.0 / pair.Value:F0}%/{parked * 100.0 / pair.Value:F0}%";
})));
Console.Error.WriteLine(
"[PERF][GUEST] top_thread: " +
string.Join(
" | ",
byThread.OrderByDescending(pair => pair.Value)
.Take(10)
.Select(pair => $"{pair.Key}={pair.Value * 100.0 / total:F1}%")));
}
/// <summary>
/// Tags a sampled address with the region it belongs to. Guest module code
/// lives above the image base; anything else is emulator or system code that
/// the managed profiler already covers.
/// </summary>
private string DescribeGuestAddress(ulong address)
{
if (address >= GuestImageBase && address < GuestImageLimit)
{
return $"(app+0x{address - GuestImageBase:X})";
}
for (var index = 0; index < _importEntries.Length; index++)
{
if (_importEntries[index].Address == (address & ~0xFUL))
{
return $"(stub:{_importEntries[index].Nid})";
}
}
return "(host)";
}
}
@@ -54,10 +54,13 @@ public sealed partial class DirectExecutionBackend
var startTicks = System.Diagnostics.Stopwatch.GetTimestamp();
var r = directExecutionBackend.DispatchImport(importIndex, argPackPtr);
RecordPerfHleDispatchTime(System.Diagnostics.Stopwatch.GetTimestamp() - startTicks);
directExecutionBackend.ClearActiveImportIndex();
return r;
}
return directExecutionBackend.DispatchImport(importIndex, argPackPtr);
var result = directExecutionBackend.DispatchImport(importIndex, argPackPtr);
directExecutionBackend.ClearActiveImportIndex();
return result;
}
catch (Exception ex)
{
@@ -69,11 +72,7 @@ public sealed partial class DirectExecutionBackend
private unsafe static int RawVectoredHandlerManaged(void* exceptionInfo)
{
EXCEPTION_RECORD* exceptionRecord = ((EXCEPTION_POINTERS*)exceptionInfo)->ExceptionRecord;
if (exceptionRecord->ExceptionCode == 3221225477u &&
exceptionRecord->NumberParameters >= 2 &&
SharpEmu.HLE.GuestImageWriteTracker.TryHandleWriteFault(
exceptionRecord->ExceptionInformation[1]))
if (TryHandleGuestImageWriteFault(exceptionInfo))
{
return -1;
}
@@ -81,6 +80,37 @@ public sealed partial class DirectExecutionBackend
return TryRecoverUnresolvedSentinel(exceptionInfo);
}
/// <summary>
/// Windows counterpart of the POSIX SIGSEGV bridge into
/// <see cref="SharpEmu.HLE.GuestImageWriteTracker"/>. Guest code runs natively,
/// so a store into a surface the GPU backend has cached is an ordinary CPU
/// write with nothing to intercept — the page is write-protected instead and
/// the resulting fault is what tells the backend to re-upload. Without this
/// the cache serves the first upload forever, and anything the guest CPU
/// draws (a software-decoded movie frame, a memset fog layer) never reaches
/// the screen.
/// </summary>
private unsafe static bool TryHandleGuestImageWriteFault(void* exceptionInfo)
{
if (!SharpEmu.HLE.GuestImageWriteTracker.Enabled)
{
return false;
}
var exceptionRecord = ((EXCEPTION_POINTERS*)exceptionInfo)->ExceptionRecord;
// STATUS_ACCESS_VIOLATION, and only the write flavour: ExceptionInformation
// is [accessKind, address] with 0=read, 1=write, 8=DEP execute.
if (exceptionRecord->ExceptionCode != 3221225477u ||
exceptionRecord->NumberParameters < 2 ||
exceptionRecord->ExceptionInformation[0] != 1uL)
{
return false;
}
return SharpEmu.HLE.GuestImageWriteTracker.TryHandleWriteFault(
exceptionRecord->ExceptionInformation[1]);
}
private unsafe static int RawUnhandledFilterManaged(void* exceptionInfo)
{
return TryRecoverUnresolvedSentinel(exceptionInfo);
@@ -174,6 +204,10 @@ public sealed partial class DirectExecutionBackend
{
RecordPerfHleCall(importStubEntry.Export?.Name ?? importStubEntry.Nid);
}
if (_profileGuestRip)
{
EnsureGuestRipSampler();
}
int num2 = Volatile.Read(in _rawSentinelRecoveries);
if (num2 != _lastReportedRawSentinelRecoveries)
{
@@ -187,6 +221,10 @@ public sealed partial class DirectExecutionBackend
}
cpuContext.Rip = importStubEntry.Address;
if (_profileGuestRip)
{
cpuContext.ActiveImportIndex = importIndex;
}
LoadImportVolatileArguments(cpuContext, argPackPtr);
cpuContext[CpuRegister.Rdi] = *(ulong*)argPackPtr;
cpuContext[CpuRegister.Rsi] = *(ulong*)(argPackPtr + 8);
@@ -1453,7 +1491,7 @@ public sealed partial class DirectExecutionBackend
string.Equals(nid, "fzyMKs9kim0", StringComparison.Ordinal) &&
result == OrbisGen2Result.ORBIS_GEN2_ERROR_TIMED_OUT;
var expectedMutexTrylockBusy =
string.Equals(nid, "K-jXhbt2gn4", StringComparison.Ordinal) &&
(nid is "K-jXhbt2gn4" or "upoVrzMHFeE") &&
result == OrbisGen2Result.ORBIS_GEN2_ERROR_BUSY;
var expectedSemaphoreTrywaitAgain =
string.Equals(nid, "H2a+IN9TP0E", StringComparison.Ordinal) &&
@@ -1470,6 +1508,9 @@ public sealed partial class DirectExecutionBackend
var expectedPrivacyInvalidParameter =
string.Equals(nid, "D-CzAxQL0XI", StringComparison.Ordinal) &&
resultValue == unchecked((int)0x80960009);
var expectedPlayGoChunkEnumerationEnd =
string.Equals(nid, "uWIYLFkkwqk", StringComparison.Ordinal) &&
resultValue == unchecked((int)0x80B2000C);
if (!expectedFileProbeMiss &&
!expectedTimedWaitTimeout &&
!expectedEqueueTimeout &&
@@ -1478,7 +1519,8 @@ public sealed partial class DirectExecutionBackend
!expectedPollSemaBusy &&
!expectedNetAcceptWouldBlock &&
!expectedUserServiceNoEvent &&
!expectedPrivacyInvalidParameter)
!expectedPrivacyInvalidParameter &&
!expectedPlayGoChunkEnumerationEnd)
{
return true;
}
@@ -5314,7 +5314,7 @@ public sealed unsafe partial class DirectExecutionBackend : INativeCpuBackend, I
var hostCpu = processorCount < 8
? guestCpu % processorCount
: processorCount >= 16
? guestCpu * 2
? MapGuestCpuAcrossSmtLanes(guestCpu, processorCount)
: guestCpu;
if (hostCpu < processorCount)
{
@@ -5325,6 +5325,45 @@ public sealed unsafe partial class DirectExecutionBackend : INativeCpuBackend, I
return hostAffinityMask;
}
/// <summary>
/// Places guest CPUs on distinct physical cores first, then wraps onto the
/// SMT siblings. Doubling the index alone only works while the title stays
/// inside the first half of the guest CPU set: beyond that every mapped lane
/// lands past the host's processor count and gets dropped, which silently
/// leaves those threads unpinned. Demon's Souls asks for CPUs 0-12 and keeps
/// its renderer on 9 and 11, so dropping the overflow un-pinned both the
/// renderer and a third of its job pool onto every core at once.
/// </summary>
private static int MapGuestCpuAcrossSmtLanes(int guestCpu, int processorCount)
{
// Reserve the top lanes for the emulator itself. A title sized for a
// console's dedicated cores will happily keep a worker per guest CPU
// spinning on an empty queue — Demon's Souls' job pool runs ~90% busy
// doing nothing — and spreading those across every host lane leaves the
// GPU translation and present threads fighting them for a slice. Packing
// near-idle spinners tighter costs them almost nothing and buys back
// whole cores for the work that actually produces frames.
var usableLanes = Math.Max(processorCount - EmulatorReservedLanes, 2);
var physicalCores = usableLanes / 2;
var lane = guestCpu % usableLanes;
return lane < physicalCores
? lane * 2
: ((lane - physicalCores) * 2) + 1;
}
/// <summary>
/// Host lanes kept away from guest threads. Measured on a 16-lane host with
/// Demon's Souls: reserving 0/4/6/8 lanes gave 6.08/6.78/7.20/5.62 fps, so
/// the useful range is a bit over a third of the machine — too few and the
/// emulator is crowded out, too many and the guest cannot make progress.
/// </summary>
private static readonly int EmulatorReservedLanes =
int.TryParse(
Environment.GetEnvironmentVariable("SHARPEMU_RESERVED_HOST_LANES"),
out var reserved) && reserved >= 0
? reserved
: Math.Max(2, Environment.ProcessorCount * 3 / 8);
public bool TrySetGuestThreadPriority(ulong guestThreadHandle, int guestPriority)
{
lock (_guestThreadGate)