mirror of
https://github.com/par274/sharpemu.git
synced 2026-07-30 22:49:53 +08:00
db4339f698
* fix(kernel): implement APR ResolveFilepathsWithPrefixToIdsAndFileSizes Resource streamers resolve relative paths against a shared prefix; without this HLE every call returned NOT_FOUND and assets never got real ids/sizes. * fix(remoteplay): stub Initialize and GetConnectionStatus as disconnected Titles probe Remote Play during pad/network bring-up; unresolved imports returned NOT_FOUND. Report initialized + disconnected so callers take the normal offline path. * fix(agc): accept Gen5 hull shaders that omit PGM_LO/HI in CreateShader Type-5 headers can start with RSRC1/RSRC2; rejecting them left null handles and Main Thread AVs. Scan the SH table and skip PGM patch when absent. Co-authored-by: Cursor <cursoragent@cursor.com> * fix(kernel): reject getdents on file fds and emit . / .. for empty dirs Returning rax=0 for non-directory or empty listings looked like EOF and let GTA treat the fd as a pointer (fiWriteAsyncDataWorker AV at 0xB1). * fix(hle): enable GuestImageWriteTracker CPU sync on Windows Windows previously hard-disabled the tracker, so CPU-written guest planes never marked dirty and host textures stayed empty. Arm pages with VirtualProtect, handle write AVs in VEH, and warm/test on VirtualAlloc memory so protect cannot poison the CRT heap. * fix(agc): skip CB metadata draws for EliminateFastClear/Fmask/DCC CB_COLOR_CONTROL modes 2/5/6 are colour-buffer metadata ops; applying the bound shader as a normal colour draw corrupts subsequent composites. Decode MODE from bits [6:4] and return before translate. * fix(agc): merge Prospero attrib-table formats onto IR vertex inputs IR-discovered BufferLoadFormat often keeps a stale float sharp format; patch DataFormat/offset from the AGC attrib table (semantic index), allow offen fetches, and map quirks 113/121 through NarrowVk for host vertex input. * fix(audio): harden AudioOut2 stack out-buffer writes against canary smash Titles that stack-allocate AudioOut2 outs next to the frame canary were corrupted by oversized or mistyped HLE writes; keep ContextPush pacing. * Revert "fix(memory): reserve only large regions (#608)" This reverts commit8f9456229a. * fix(gpu): decode Gen5 R16 and RG32 render-target formats * fix(audio): AudioOut2 host beds, deeper waveOut queue, AJM MP3 GTA V Enhanced routes intro/menu audio through AudioOut2 and FMOD's AJM MP3 path. Wire PortCreate/PortSetAttributes/ContextPush to dual host stereo streams, deepen WinMM queue to 128KiB, and decode AJM codec 0 with a stateful NLayer helper so menu music is not silent. * fix(agc): map PS interpolants via SPI_PS_INPUT_CNTL semantics Identity ATTR→param wiring ignored hardware remapping, so UI draws got wrong (or empty) interpolants. Pack CNTL from matched PS/GS semantics, thread it into Vulkan/Metal as Location/Flat, and fingerprint it in the graphics shader cache key. * fix(agc): rect-list/NGG strips, Index8 expand, and GE_INDX_OFFSET NGG single-rect UI needs triangle-strip expansion; Prospero Index8 must expand to host u16; glyphs need base vertex from GE_INDX_OFFSET. Skip param-less rect-lists instead of inventing colour draws. * fix(np): report GTA Story Mode addcont entitlements as owned NpEntitlementAccess was returning an empty add-on list, so GTA V Enhanced offered Buy Story Mode. Publish the installed license labels and stub premium-event registration so offline sessions take the owned path. * fix(cpu): prefer native workers for all guest entry stubs Route thread entry, continuation, and main entry through RunGuestEntryStub so guest stubs are not invoked above CLR-managed frames (UnmanagedCallersOnly FailFast). Keep requireNativeWorker for tbb_thead; other paths prefer workers with calli fallback. * fix(agc): implement Rewind/Jump writers and IT_REWIND waits GTA Subrender AVs came from AcbJumpGetSize / DcbRewind returning NOT_FOUND as packet sizes. Add IT_REWIND and INDIRECT_BUFFER writers, patch SetRewindState into the GPU wait registry, and nest-parse 4-dword jumps. * fix(gpu): use AddrLib ExactXor for Gen5 Standard256B (mode 1) Mode 5 already had Standard4K ExactXor; mode 1 still used the generic StandardSwizzle block table, which mis-detiles Gen5 UI atlases. * Revert "fix(cpu): prefer native workers for all guest entry stubs" This reverts commit31c4db0d38. * fix(memory): commit-first large maps; reserve only on failure Replace the #608 always-reserve-only exact-map path with allocate-first and lazy reserve fallback when a huge non-exec commit cannot be satisfied. Prime and widen GetPointer commit so the fallback path is safer for native walkers. Drops the need for a hard #608 revert. * [Agc] Implement fused shader half exports * fix(agc): accept optional hull state in CreatePrimState Port the CreatePrimState hull-optional path from #583 so fused HS pipelines (GTA) are not rejected with INVALID_ARGUMENT. Geometry-derived CX/UC writes are unchanged; hull is traced only. * fix(videoout): restore thread-safe VulkanHostBufferPool (#564) The6db095ewipe dropped CasualcoderDev's lock-ordering-safe pool. Concurrent Return/TryTake without the gate races after the first present and can hang the submit path. * Revert "fix(agc): implement Rewind/Jump writers and IT_REWIND waits" This reverts commitbec77bf083. * test(memory): align lazy-commit expectations with commit-first policy Fake hosts must reject Allocate so reserve-only paths still run, and GetPointer asserts the 32 MiB prime range including AlignUp spill. * diag(gpu): log guest-queue backlog breakdown under backpressure Rate-limit top work types and ordered debugName prefixes when the Vulkan guest work queue stalls, so North Yankton logs show acquire/label vs draw traffic instead of only VulkanOrderedGuestAction. * perf(agc): coalesce acquire flushes and batch non-DMA label wakes Flush pending ACQUIRE_MEM invalidation at draw/dispatch/dma/flip boundaries instead of before every packet, and complete release/write-data producers in the same ordered action so load paths enqueue far fewer VulkanOrderedGuestAction items. * perf(gpu): wait for ordered-action fences and keep draining sync On Windows/Linux, block briefly for queue-visibility fences instead of deferring the whole logical queue for the tick. Prefer ordered sync/flip heads under backlog pressure, and keep macOS non-blocking defer behavior. * perf(gpu): raise sync-item ceiling above payload guest-work cap Apply SHARPEMU_PENDING_GUEST_WORK_ITEMS mainly to compute/draw/image payload work, and allow a higher SHARPEMU_PENDING_GUEST_SYNC_ITEMS ceiling for zero-payload ordered actions and flip markers. Keep the byte budget as the RAM safety valve. * fix(gta): stub Voice ports and implement sceKernelCheckReachability Resolve North Yankton-path Voice Create/Delete/Connect/Disconnect/End NIDs and EnumerationThread reachability checks so leftover unresolved imports are not on the critical path. * diag(gta): arm flip/present/wait probes after North Audio Rate-limited load_progress TRACE for flip submit, ordered flip enqueue, present taken/not-taken, and GPU wait backlog so North Yankton freezes can be classified without full AGC tracing. * fix(ampr): restore sequential offset=-1 reads for streamer packs Re-wire PakDirectoryTracker into sceAmprAprCommandBufferReadFile (dropped in #216) so RAGE sequential pack reads no longer fail while the North Yankton UI keeps flipping. Also rate-limit CheckReachability miss paths for EnumerationThread diagnosis. * fix(hle/videoout): Windows GuestImage opt-in and keep GTA intro without sync Default the tracker off on Windows to avoid VirtualProtect thrash, gate AGC texel-copy skips on Enabled so guest Bink planes keep shipping pixels, and drain CPU-written images on the present thread when sync is opted in. * fix(videoout): probe guest content when tracker off so UI can skip copies Restores upload-known/texture-cache skips for Dead Cells menus, and uses a sparse guest-memory fingerprint when GuestImageWriteTracker is disabled so CPU-updated Bink planes still force texel copies for GTA intro. * fix(audio): keep 128KiB host queue AudioOut2-only Restore the default 32 KiB (~171 ms) PCM bed for classic AudioOut so titles like Dreaming Sarah stay in sync; only AudioOut2 opens the deeper queue needed for bursty FMOD Push on GTA. --------- Co-authored-by: samto6 <123419830+samto6@users.noreply.github.com>
294 lines
11 KiB
C#
294 lines
11 KiB
C#
// Copyright (C) 2026 SharpEmu Emulator Project
|
|
// SPDX-License-Identifier: GPL-2.0-or-later
|
|
|
|
using System.Buffers.Binary;
|
|
using SharpEmu.HLE;
|
|
using SharpEmu.Libs.Agc;
|
|
using SharpEmu.ShaderCompiler;
|
|
using Xunit;
|
|
|
|
namespace SharpEmu.Libs.Tests.Agc;
|
|
|
|
/// <summary>
|
|
/// Coverage for AGC attrib-table → BufferFormat merge and semantic indexing.
|
|
/// </summary>
|
|
public sealed class AgcVertexMetadataTests
|
|
{
|
|
[Fact]
|
|
public void BuildVertexResources_UsesSemanticNotHardwareMappingAsAttribIndex()
|
|
{
|
|
// input_semantics[0]: semantic=1, hardware_mapping=4, size=2
|
|
// If hardware_mapping were wrongly used as the attrib index, we'd read
|
|
// attrib[4] instead of attrib[1] and get the wrong format/offset.
|
|
const ulong memoryBase = 0x1_0000_0000;
|
|
var memory = new FakeCpuMemory(memoryBase, 0x2000);
|
|
var ctx = new CpuContext(memory, Generation.Gen5);
|
|
|
|
const ulong semanticsAddress = memoryBase + 0x100;
|
|
const ulong attribTable = memoryBase + 0x200;
|
|
const ulong bufferTable = memoryBase + 0x300;
|
|
const ulong sharpBase = memoryBase + 0x800;
|
|
|
|
// ShaderSemantic word: semantic=1, hw_mapping=4, size_in_elements=2
|
|
WriteUInt32(memory, semanticsAddress, 1u | (4u << 8) | (2u << 16));
|
|
|
|
// attrib[0] unused garbage
|
|
WriteUInt32(memory, attribTable, 0xDEAD_BEEFu);
|
|
// attrib[1]: buffer=0, format=k16_16Float(29), offset=8, fetch=0
|
|
WriteUInt32(memory, attribTable + 4, 0u | (29u << 5) | (8u << 14));
|
|
|
|
// V# at buffer table[0]: base=sharpBase, stride=16
|
|
WriteUInt32(memory, bufferTable, (uint)(sharpBase & 0xFFFF_FFFFUL));
|
|
WriteUInt32(
|
|
memory,
|
|
bufferTable + 4,
|
|
(uint)(sharpBase >> 32) | (16u << 16));
|
|
|
|
var scalars = new uint[32];
|
|
scalars[8] = (uint)(attribTable & 0xFFFF_FFFFUL);
|
|
scalars[9] = (uint)(attribTable >> 32);
|
|
scalars[10] = (uint)(bufferTable & 0xFFFF_FFFFUL);
|
|
scalars[11] = (uint)(bufferTable >> 32);
|
|
|
|
var tables = new AgcVertexMetadata.VertexTableRegisters(
|
|
VertexBufferReg: 10,
|
|
VertexAttribReg: 8,
|
|
InputSemanticsCount: 1,
|
|
InputSemanticsAddress: semanticsAddress);
|
|
|
|
Assert.True(
|
|
AgcVertexMetadata.TryBuildVertexResourcesFromMetadata(
|
|
ctx,
|
|
scalars,
|
|
tables,
|
|
out var resources));
|
|
Assert.Single(resources);
|
|
Assert.Equal(1u, resources[0].Semantic);
|
|
Assert.Equal(4u, resources[0].HardwareMapping);
|
|
Assert.Equal(8u, resources[0].OffsetBytes);
|
|
Assert.Equal(5u, resources[0].DataFormat); // R16G16
|
|
Assert.Equal(7u, resources[0].NumberFormat); // Float
|
|
Assert.Equal(2u, resources[0].ComponentCount);
|
|
Assert.Equal(sharpBase, resources[0].SharpBase);
|
|
Assert.False(resources[0].PerInstance);
|
|
}
|
|
|
|
[Fact]
|
|
public void MergeVertexInputs_OverlaysFormatWithoutRebasingCapture()
|
|
{
|
|
const ulong memoryBase = 0x1_0000_0000;
|
|
var memory = new FakeCpuMemory(memoryBase, 0x2000);
|
|
var ctx = new CpuContext(memory, Generation.Gen5);
|
|
|
|
const ulong semanticsAddress = memoryBase + 0x100;
|
|
const ulong attribTable = memoryBase + 0x200;
|
|
const ulong bufferTable = memoryBase + 0x300;
|
|
const ulong sharpBase = memoryBase + 0x800;
|
|
|
|
WriteUInt32(memory, semanticsAddress, 0u | (0u << 8) | (4u << 16));
|
|
// format k8_8_8_8UNorm(56), offset=12
|
|
WriteUInt32(memory, attribTable, 0u | (56u << 5) | (12u << 14));
|
|
WriteUInt32(memory, bufferTable, (uint)(sharpBase & 0xFFFF_FFFFUL));
|
|
WriteUInt32(memory, bufferTable + 4, (uint)(sharpBase >> 32) | (16u << 16));
|
|
|
|
var scalars = new uint[32];
|
|
scalars[4] = (uint)(attribTable & 0xFFFF_FFFFUL);
|
|
scalars[5] = (uint)(attribTable >> 32);
|
|
scalars[6] = (uint)(bufferTable & 0xFFFF_FFFFUL);
|
|
scalars[7] = (uint)(bufferTable >> 32);
|
|
|
|
var tables = new AgcVertexMetadata.VertexTableRegisters(
|
|
VertexBufferReg: 6,
|
|
VertexAttribReg: 4,
|
|
InputSemanticsCount: 1,
|
|
InputSemanticsAddress: semanticsAddress);
|
|
|
|
var data = new byte[64];
|
|
var discovered = new[]
|
|
{
|
|
new Gen5VertexInputBinding(
|
|
Pc: 0x40,
|
|
Location: 0,
|
|
ComponentCount: 4,
|
|
DataFormat: 14, // wrong IR guess
|
|
NumberFormat: 7,
|
|
BaseAddress: sharpBase,
|
|
Stride: 16,
|
|
OffsetBytes: 0,
|
|
Data: data,
|
|
DataLength: data.Length,
|
|
DataPooled: false),
|
|
};
|
|
|
|
var merged = AgcVertexMetadata.MergeVertexInputsFromMetadata(
|
|
ctx,
|
|
scalars,
|
|
tables,
|
|
discovered);
|
|
Assert.Single(merged);
|
|
Assert.Equal(0u, merged[0].Location);
|
|
Assert.Equal(sharpBase, merged[0].BaseAddress);
|
|
Assert.Same(data, merged[0].Data);
|
|
Assert.Equal(10u, merged[0].DataFormat); // RGBA8
|
|
Assert.Equal(0u, merged[0].NumberFormat); // Unorm
|
|
Assert.Equal(12u, merged[0].OffsetBytes);
|
|
Assert.Equal(0x40u, merged[0].Pc);
|
|
}
|
|
|
|
[Fact]
|
|
public void MergeVertexInputs_AcceptsVertexAttribFormatEnums()
|
|
{
|
|
// Attrib tables store VertexAttribFormat (227 = rgba8 unorm), not
|
|
// BufferFormat (56). Without conversion the format patch is a no-op.
|
|
const ulong memoryBase = 0x1_0000_0000;
|
|
var memory = new FakeCpuMemory(memoryBase, 0x2000);
|
|
var ctx = new CpuContext(memory, Generation.Gen5);
|
|
|
|
const ulong semanticsAddress = memoryBase + 0x100;
|
|
const ulong attribTable = memoryBase + 0x200;
|
|
const ulong bufferTable = memoryBase + 0x300;
|
|
const ulong sharpBase = memoryBase + 0x800;
|
|
|
|
WriteUInt32(memory, semanticsAddress, 0u | (0u << 8) | (4u << 16));
|
|
WriteUInt32(memory, attribTable, 0u | (227u << 5) | (12u << 14)); // VertexAttribFormat
|
|
WriteUInt32(memory, bufferTable, (uint)(sharpBase & 0xFFFF_FFFFUL));
|
|
WriteUInt32(memory, bufferTable + 4, (uint)(sharpBase >> 32) | (16u << 16));
|
|
|
|
var scalars = new uint[32];
|
|
scalars[4] = (uint)(attribTable & 0xFFFF_FFFFUL);
|
|
scalars[5] = (uint)(attribTable >> 32);
|
|
scalars[6] = (uint)(bufferTable & 0xFFFF_FFFFUL);
|
|
scalars[7] = (uint)(bufferTable >> 32);
|
|
|
|
var tables = new AgcVertexMetadata.VertexTableRegisters(
|
|
VertexBufferReg: 6,
|
|
VertexAttribReg: 4,
|
|
InputSemanticsCount: 1,
|
|
InputSemanticsAddress: semanticsAddress);
|
|
|
|
var data = new byte[64];
|
|
var discovered = new[]
|
|
{
|
|
new Gen5VertexInputBinding(
|
|
0x40, 0, 4, 14, 7, sharpBase, 16, 12, data, data.Length, false),
|
|
};
|
|
|
|
var merged = AgcVertexMetadata.MergeVertexInputsFromMetadata(
|
|
ctx,
|
|
scalars,
|
|
tables,
|
|
discovered);
|
|
Assert.Equal(10u, merged[0].DataFormat);
|
|
Assert.Equal(0u, merged[0].NumberFormat);
|
|
Assert.Equal(12u, merged[0].OffsetBytes);
|
|
}
|
|
|
|
[Fact]
|
|
public void MergeVertexInputs_MatchesInterleavedAttrsByOffsetNotBareBase()
|
|
{
|
|
// Both attributes share SharpBase. Matching by base alone would assign
|
|
// the color format to position (video/UI regression).
|
|
const ulong memoryBase = 0x1_0000_0000;
|
|
var memory = new FakeCpuMemory(memoryBase, 0x2000);
|
|
var ctx = new CpuContext(memory, Generation.Gen5);
|
|
|
|
const ulong semanticsAddress = memoryBase + 0x100;
|
|
const ulong attribTable = memoryBase + 0x200;
|
|
const ulong bufferTable = memoryBase + 0x300;
|
|
const ulong sharpBase = memoryBase + 0x800;
|
|
|
|
// semantic0 → pos float4 @0; semantic1 → color rgba8 @12
|
|
WriteUInt32(memory, semanticsAddress, 0u | (0u << 8) | (4u << 16));
|
|
WriteUInt32(memory, semanticsAddress + 4, 1u | (4u << 8) | (4u << 16));
|
|
WriteUInt32(memory, attribTable, 0u | (77u << 5) | (0u << 14)); // k32_32_32_32Float
|
|
WriteUInt32(memory, attribTable + 4, 0u | (56u << 5) | (12u << 14)); // rgba8unorm @12
|
|
WriteUInt32(memory, bufferTable, (uint)(sharpBase & 0xFFFF_FFFFUL));
|
|
WriteUInt32(memory, bufferTable + 4, (uint)(sharpBase >> 32) | (16u << 16));
|
|
|
|
var scalars = new uint[32];
|
|
scalars[4] = (uint)(attribTable & 0xFFFF_FFFFUL);
|
|
scalars[5] = (uint)(attribTable >> 32);
|
|
scalars[6] = (uint)(bufferTable & 0xFFFF_FFFFUL);
|
|
scalars[7] = (uint)(bufferTable >> 32);
|
|
|
|
var tables = new AgcVertexMetadata.VertexTableRegisters(
|
|
VertexBufferReg: 6,
|
|
VertexAttribReg: 4,
|
|
InputSemanticsCount: 2,
|
|
InputSemanticsAddress: semanticsAddress);
|
|
|
|
var data = new byte[64];
|
|
var discovered = new[]
|
|
{
|
|
new Gen5VertexInputBinding(
|
|
0x40, 0, 4, 14, 7, sharpBase, 16, 0, data, data.Length, false),
|
|
new Gen5VertexInputBinding(
|
|
0x80, 1, 4, 14, 7, sharpBase, 16, 12, data, data.Length, false),
|
|
};
|
|
|
|
var merged = AgcVertexMetadata.MergeVertexInputsFromMetadata(
|
|
ctx,
|
|
scalars,
|
|
tables,
|
|
discovered);
|
|
Assert.Equal(2, merged.Count);
|
|
Assert.Equal(0u, merged[0].OffsetBytes);
|
|
Assert.Equal(12u, merged[1].OffsetBytes);
|
|
Assert.Equal(0u, merged[1].NumberFormat); // Unorm color, not float
|
|
Assert.Equal(10u, merged[1].DataFormat); // RGBA8
|
|
Assert.Equal(sharpBase, merged[0].BaseAddress);
|
|
Assert.Equal(sharpBase, merged[1].BaseAddress);
|
|
Assert.Same(data, merged[0].Data);
|
|
}
|
|
|
|
[Fact]
|
|
public void CollectFetchPrologPcs_FindsSBufferLoadsFromTableRegisters()
|
|
{
|
|
var tables = new AgcVertexMetadata.VertexTableRegisters(
|
|
VertexBufferReg: 10,
|
|
VertexAttribReg: 8,
|
|
InputSemanticsCount: 1,
|
|
InputSemanticsAddress: 1);
|
|
|
|
var program = new Gen5ShaderProgram(
|
|
0,
|
|
[
|
|
new Gen5ShaderInstruction(
|
|
0x10,
|
|
Gen5ShaderEncoding.Smem,
|
|
"SBufferLoadDword",
|
|
Words: [],
|
|
Sources: [Gen5Operand.Scalar(8)],
|
|
Destinations: [Gen5Operand.Scalar(20)],
|
|
new Gen5ScalarMemoryControl(1, 0, null)),
|
|
new Gen5ShaderInstruction(
|
|
0x20,
|
|
Gen5ShaderEncoding.Smem,
|
|
"SBufferLoadDword",
|
|
Words: [],
|
|
Sources: [Gen5Operand.Scalar(12)],
|
|
Destinations: [Gen5Operand.Scalar(24)],
|
|
new Gen5ScalarMemoryControl(1, 0, null)),
|
|
new Gen5ShaderInstruction(
|
|
0x30,
|
|
Gen5ShaderEncoding.Sopp,
|
|
"SEndpgm",
|
|
Words: [],
|
|
Sources: [],
|
|
Destinations: [],
|
|
null),
|
|
]);
|
|
|
|
var pcs = AgcVertexMetadata.CollectFetchPrologPcs(program, tables);
|
|
Assert.Contains(0x10u, pcs);
|
|
Assert.DoesNotContain(0x20u, pcs);
|
|
}
|
|
|
|
private static void WriteUInt32(FakeCpuMemory memory, ulong address, uint value)
|
|
{
|
|
Span<byte> bytes = stackalloc byte[4];
|
|
BinaryPrimitives.WriteUInt32LittleEndian(bytes, value);
|
|
Assert.True(memory.TryWrite(address, bytes));
|
|
}
|
|
}
|