[AGC] Fix arena/fence tracking stalls and recover lost GPU progress (#770)

* [AGC] Fix arena/fence tracking stalls and recover lost GPU progress

Fix several AGC synchronization issues that could leave GPU work
unsubmitted or fences permanently unsatisfied:

- Recover fence writes missed during arena transitions by tracking
  closed arena tails and resweeping unresolved fence locations.
- Fix chunk submission accounting so only actually parsed ranges are
  considered submitted, avoiding missed orphan recovery.
- Extend arena sweeps to follow chained command ranges without
  duplicate submissions.
- Preserve GPU state tracking across native worker memory wrappers by
  using canonical guest memory identity.
- Fix builder-ring orphan submissions and stale ring-tail waits.
- Improve frame cycling by handling arena reuse, cursor regression,
  and transiently unavailable builder headers.
- Record conditional DCB execution (IT_COND_EXEC) instead of
  silently dropping the packet.

These changes improve AGC progress tracking and prevent GPU fence
starvation/deadlocks on titles relying on builder arenas and indirect
command chains.

* chore: trigger CI
This commit is contained in:
Foued Attar
2026-08-07 13:37:09 +02:00
committed by GitHub
parent 418eb7ecea
commit 62c3852556
4 changed files with 1821 additions and 76 deletions
+3 -1
View File
@@ -42,4 +42,6 @@ ehthumbs.db
.vs/
.idea/
.vscode/
.vscode/
clean_test/
.debug/
File diff suppressed because it is too large Load Diff
+84 -1
View File
@@ -60,6 +60,19 @@ internal static class GpuWaitRegistry
// address) so distinct guest processes never alias.
private static readonly Dictionary<(object, ulong), ulong> _lastProduced = new();
// Unwraps to the shared root: per-thread TrackedCpuMemory decorators
// over ONE virtual memory are not reference-equal, so a raw-reference
// filter would make waits invisible across threads.
private static object? Canonicalize(object? memory)
{
while (memory is SharpEmu.HLE.ICpuMemoryWrapper wrapper)
{
memory = wrapper.Inner;
}
return memory;
}
public static int Count
{
get
@@ -79,6 +92,7 @@ internal static class GpuWaitRegistry
public static int CountForMemory(object memory)
{
memory = Canonicalize(memory)!;
lock (_gate)
{
var total = 0;
@@ -155,6 +169,7 @@ internal static class GpuWaitRegistry
public static void Register(ulong address, WaitingDcb waiter)
{
waiter.WaitAddress = address;
waiter.Memory = Canonicalize(waiter.Memory);
lock (_gate)
{
if (!_waiters.TryGetValue(address, out var list))
@@ -177,6 +192,7 @@ internal static class GpuWaitRegistry
object memory,
Func<ulong, bool, ulong?> readValue)
{
memory = Canonicalize(memory)!;
List<WaitingDcb>? woken = null;
lock (_gate)
{
@@ -237,6 +253,7 @@ internal static class GpuWaitRegistry
long nowTicks,
long maxAgeTicks)
{
memory = Canonicalize(memory)!;
List<WaitingDcb>? stale = null;
lock (_gate)
{
@@ -273,6 +290,7 @@ internal static class GpuWaitRegistry
ulong start,
ulong length)
{
memory = Canonicalize(memory)!;
var matches = new List<(ulong Address, int Count)>();
if (length == 0)
{
@@ -355,6 +373,56 @@ internal static class GpuWaitRegistry
return latchedAny;
}
/// <summary>
/// Every registered waiter, for the flip-stall watchdog. Not filtered by
/// memory identity — the watchdog wants a whole-process view.
/// </summary>
public static List<WaitingDcb> SnapshotAll()
{
var snapshot = new List<WaitingDcb>();
lock (_gate)
{
foreach (var (_, list) in _waiters)
{
snapshot.AddRange(list);
}
}
return snapshot;
}
/// <summary>
/// Removes the waiter at <paramref name="address"/> whose State is
/// <paramref name="state"/> — used when a new submission supersedes a
/// ring-tail park that would otherwise pin the queue forever.
/// </summary>
public static bool TryRemoveByState(object state, ulong address)
{
lock (_gate)
{
if (!_waiters.TryGetValue(address, out var list))
{
return false;
}
for (var i = list.Count - 1; i >= 0; i--)
{
if (ReferenceEquals(list[i].State, state))
{
list.RemoveAt(i);
if (list.Count == 0)
{
_waiters.Remove(address);
}
return true;
}
}
return false;
}
}
/// <summary>
/// Removes and returns waiters carrying a <see cref="WaitingDcb.RetryDeadlineTicks"/>
/// that has elapsed. Used for indirect-dispatch dimension retries: the caller
@@ -564,6 +632,21 @@ internal static class GpuWaitRegistry
return broken;
}
// Under orphan force-submit, producers can run ahead of waiter
// registration and pass an equal-compare value before it's ever seen.
// Treat == as "reached or passed" only in that mode, so other titles
// keep exact hardware semantics. SHARPEMU_GPU_WAIT_EQ_EXACT=1 restores
// strict equality for A/B.
private static readonly bool _equalCompareExact =
string.Equals(
Environment.GetEnvironmentVariable("SHARPEMU_GPU_WAIT_EQ_EXACT"),
"1",
StringComparison.Ordinal) ||
!string.Equals(
Environment.GetEnvironmentVariable("SHARPEMU_FORCE_SUBMIT_ORPHAN_PREAMBLES"),
"1",
StringComparison.Ordinal);
public static bool Compare(in WaitingDcb waiter, ulong value)
{
var masked = value & waiter.Mask;
@@ -573,7 +656,7 @@ internal static class GpuWaitRegistry
0 => true,
1 => masked < reference,
2 => masked <= reference,
3 => masked == reference,
3 => _equalCompareExact ? masked == reference : masked >= reference,
4 => masked != reference,
5 => masked >= reference,
6 => masked > reference,
+1
View File
@@ -24,6 +24,7 @@ SPDX-License-Identifier: GPL-2.0-or-later
<ItemGroup>
<InternalsVisibleTo Include="SharpEmu.Libs.Tests" />
<!-- SharpEmu.Core's stall watchdog reads GpuWaitRegistry for diagnostics. -->
<InternalsVisibleTo Include="SharpEmu.Core" />
</ItemGroup>