diff --git a/ps2xRecomp/src/lib/fpu_translator.cpp b/ps2xRecomp/src/lib/fpu_translator.cpp index 944cee796..435db3aa9 100644 --- a/ps2xRecomp/src/lib/fpu_translator.cpp +++ b/ps2xRecomp/src/lib/fpu_translator.cpp @@ -53,10 +53,10 @@ namespace ps2recomp case COP1_S_MUL: return fmt::format("ctx->f[{}] = FPU_MUL_S(ctx->f[{}], ctx->f[{}]);", fd, fs, ft); case COP1_S_DIV: - return fmt::format("if (ctx->f[{}] == 0.0f) {{ ctx->fcr31 |= 0x100000; /* DZ flag */ " - "ctx->f[{}] = copysignf(INFINITY, ctx->f[{}] * 0.0f); }} " - "else ctx->f[{}] = ctx->f[{}] / ctx->f[{}];", - ft, fd, fs, fd, fs, ft); + // The EE has no Inf/NaN: x/0 gives +-max (sign of fs xor ft), see ps2_fdiv. + return fmt::format("{{ if (ctx->f[{}] == 0.0f) ctx->fcr31 |= 0x100000; /* DZ flag */ " + "ctx->f[{}] = FPU_DIV_S(ctx->f[{}], ctx->f[{}]); }}", + ft, fd, fs, ft); case COP1_S_SQRT: return fmt::format("ctx->f[{}] = FPU_SQRT_S(ctx->f[{}]);", fd, fs); case COP1_S_ABS: @@ -76,7 +76,8 @@ namespace ps2recomp case COP1_S_CVT_W: return fmt::format("{{ int32_t tmp = FPU_CVT_W_S(ctx->f[{}]); std::memcpy(&ctx->f[{}], &tmp, sizeof(tmp)); }}", fs, fd); case COP1_S_RSQRT: - return fmt::format("ctx->f[{}] = 1.0f / sqrtf(ctx->f[{}]);", fd, fs); + // RSQRT.S fd, fs, ft computes fs / sqrt(ft), not 1 / sqrt(fs). + return fmt::format("ctx->f[{}] = FPU_RSQRT_S(ctx->f[{}], ctx->f[{}]);", fd, fs, ft); case COP1_S_ADDA: return fmt::format("FPU_SET_ACC(ctx, FPU_ADD_S(ctx->f[{}], ctx->f[{}]));", fs, ft); case COP1_S_SUBA: diff --git a/ps2xRuntime/include/ps2_runtime_macros.h b/ps2xRuntime/include/ps2_runtime_macros.h index 0f19c9f24..cbcf9690e 100644 --- a/ps2xRuntime/include/ps2_runtime_macros.h +++ b/ps2xRuntime/include/ps2_runtime_macros.h @@ -140,10 +140,60 @@ static inline uint32_t ps2_plzcw32(uint32_t x) #define PS2_PXOR(a, b) _mm_xor_si128((__m128i)(a), (__m128i)(b)) #define PS2_PNOR(a, b) _mm_xor_si128(_mm_or_si128((__m128i)(a), (__m128i)(b)), _mm_set1_epi32(0xFFFFFFFF)) +// PS2 float semantics (EE FPU and VU). The PS2 has no Inf/NaN: a value with exponent 255 behaves like a huge +// normal number, an overflowing result saturates to +-max (0x7F7FFFFF) and x/0 gives +-max with the sign of +// (numerator xor denominator). Hosts that follow IEEE produce Inf/NaN here, which then poisons later maths +// (e.g. a normalisation of a zero-length vector). +inline float ps2_fclamp(float x) +{ + uint32_t b; + std::memcpy(&b, &x, sizeof(b)); + if ((b & 0x7F800000u) == 0x7F800000u) + b = (b & 0x80000000u) | 0x7F7FFFFFu; + std::memcpy(&x, &b, sizeof(b)); + return x; +} + +// Sign is sign(a) xor sign(b); used for a division by zero. +inline float ps2_fmax_signed(float a, float b) +{ + uint32_t ua, ub; + std::memcpy(&ua, &a, sizeof(ua)); + std::memcpy(&ub, &b, sizeof(ub)); + const uint32_t r = ((ua ^ ub) & 0x80000000u) | 0x7F7FFFFFu; + float f; + std::memcpy(&f, &r, sizeof(f)); + return f; +} + +inline float ps2_fdiv(float a, float b) +{ + a = ps2_fclamp(a); + b = ps2_fclamp(b); + return (b == 0.0f) ? ps2_fmax_signed(a, b) : ps2_fclamp(a / b); +} + +// fs / sqrt(|ft|); the radicand's sign is ignored. +inline float ps2_frsqrt(float a, float b) +{ + a = ps2_fclamp(a); + b = ps2_fclamp(b); + return (b == 0.0f) ? ps2_fmax_signed(a, b) : ps2_fclamp(a / std::sqrt(std::fabs(b))); +} + +inline __m128 ps2_vclamp(__m128 v) +{ + const __m128i bits = _mm_castps_si128(v); + const __m128i expMask = _mm_set1_epi32(0x7F800000); + const __m128i special = _mm_cmpeq_epi32(_mm_and_si128(bits, expMask), expMask); + const __m128i saturated = _mm_or_si128(_mm_and_si128(bits, _mm_set1_epi32((int)0x80000000)), _mm_set1_epi32(0x7F7FFFFF)); + return _mm_castsi128_ps(_mm_or_si128(_mm_and_si128(special, saturated), _mm_andnot_si128(special, bits))); +} + // PS2 VU (Vector Unit) operations -#define PS2_VADD(a, b) _mm_add_ps((__m128)(a), (__m128)(b)) -#define PS2_VSUB(a, b) _mm_sub_ps((__m128)(a), (__m128)(b)) -#define PS2_VMUL(a, b) _mm_mul_ps((__m128)(a), (__m128)(b)) +#define PS2_VADD(a, b) ps2_vclamp(_mm_add_ps(ps2_vclamp((__m128)(a)), ps2_vclamp((__m128)(b)))) +#define PS2_VSUB(a, b) ps2_vclamp(_mm_sub_ps(ps2_vclamp((__m128)(a)), ps2_vclamp((__m128)(b)))) +#define PS2_VMUL(a, b) ps2_vclamp(_mm_mul_ps(ps2_vclamp((__m128)(a)), ps2_vclamp((__m128)(b)))) #define PS2_VDIV(a, b) _mm_div_ps((__m128)(a), (__m128)(b)) #define PS2_VMULQ(a, q) _mm_mul_ps((__m128)(a), _mm_set1_ps(q)) #define PS2_VBLEND(a, b, mask) PS2_BLENDV_PS((__m128)(a), (__m128)(b), (__m128)(mask)) @@ -605,11 +655,12 @@ inline __m128i ps2_u64_to_epi64_pair(uint64_t value) // FPU (COP1) operations #define FPU_SET_ACC(ctx, res) (ctx->f_acc = res) -#define FPU_ADD_S(a, b) ((float)(a) + (float)(b)) -#define FPU_SUB_S(a, b) ((float)(a) - (float)(b)) -#define FPU_MUL_S(a, b) ((float)(a) * (float)(b)) -#define FPU_DIV_S(a, b) ((float)(a) / (float)(b)) -#define FPU_SQRT_S(a) sqrtf((float)(a)) +#define FPU_ADD_S(a, b) ps2_fclamp(ps2_fclamp((float)(a)) + ps2_fclamp((float)(b))) +#define FPU_SUB_S(a, b) ps2_fclamp(ps2_fclamp((float)(a)) - ps2_fclamp((float)(b))) +#define FPU_MUL_S(a, b) ps2_fclamp(ps2_fclamp((float)(a)) * ps2_fclamp((float)(b))) +#define FPU_DIV_S(a, b) ps2_fdiv((float)(a), (float)(b)) +#define FPU_RSQRT_S(a, b) ps2_frsqrt((float)(a), (float)(b)) +#define FPU_SQRT_S(a) std::sqrt(std::fabs(ps2_fclamp((float)(a)))) #define FPU_ABS_S(a) fabsf((float)(a)) #define FPU_MOV_S(a) ((float)(a)) #define FPU_NEG_S(a) (-(float)(a)) diff --git a/ps2xTest/CMakeLists.txt b/ps2xTest/CMakeLists.txt index 5db04b923..3eaa8b3d2 100644 --- a/ps2xTest/CMakeLists.txt +++ b/ps2xTest/CMakeLists.txt @@ -37,6 +37,7 @@ add_library(ps2_test_lib STATIC src/ps2_sif_dma_tests.cpp src/ps2_recompiler_tests.cpp src/ps2_runtime_expansion_tests.cpp + src/ps2_float_semantics_tests.cpp ) target_include_directories(ps2_test_lib PRIVATE diff --git a/ps2xTest/src/main.cpp b/ps2xTest/src/main.cpp index 40a68b16b..7b7889148 100644 --- a/ps2xTest/src/main.cpp +++ b/ps2xTest/src/main.cpp @@ -18,6 +18,7 @@ void register_ps2_sif_rpc_tests(); void register_ps2_sif_dma_tests(); void register_ps2_recompiler_tests(); void register_ps2_runtime_expansion_tests(); +void register_ps2_float_semantics_tests(); void reset_ps2_test_function_table(); int main() @@ -40,6 +41,7 @@ int main() register_ps2_sif_dma_tests(); register_ps2_recompiler_tests(); register_ps2_runtime_expansion_tests(); + register_ps2_float_semantics_tests(); int res = MiniTest::Run(); std::cout.flush(); std::cerr.flush(); diff --git a/ps2xTest/src/ps2_float_semantics_tests.cpp b/ps2xTest/src/ps2_float_semantics_tests.cpp new file mode 100644 index 000000000..adfec5852 --- /dev/null +++ b/ps2xTest/src/ps2_float_semantics_tests.cpp @@ -0,0 +1,157 @@ +#include "MiniTest.h" +#include "ps2recomp/code_generator.h" +#include "ps2recomp/instructions.h" +#include "ps2recomp/types.h" +#include "ps2_runtime_macros.h" + +#include +#include +#include +#include +#include + +using namespace ps2recomp; + +namespace +{ + constexpr uint32_t kPlusMax = 0x7F7FFFFFu; + constexpr uint32_t kMinusMax = 0xFF7FFFFFu; + + uint32_t bitsOf(float f) + { + uint32_t b; + std::memcpy(&b, &f, sizeof(b)); + return b; + } + + float floatOf(uint32_t b) + { + float f; + std::memcpy(&f, &b, sizeof(f)); + return f; + } + + Instruction makeCop1S(uint32_t function, uint8_t fd, uint8_t fs, uint8_t ft) + { + Instruction inst{}; + inst.opcode = OPCODE_COP1; + inst.rs = COP1_S; + inst.rt = ft; + inst.rd = fs; + inst.sa = fd; + inst.function = function; + return inst; + } + + uint32_t laneBits(__m128 v, int lane) + { + float lanes[4]; + _mm_storeu_ps(lanes, v); + return bitsOf(lanes[lane]); + } +} + +void register_ps2_float_semantics_tests() +{ + MiniTest::Case("PS2FloatSemantics", [](TestCase &tc) + { + tc.Run("ps2_fclamp turns Inf and NaN into +-max and leaves finite values alone", [](TestCase &t) + { + t.Equals(bitsOf(ps2_fclamp(std::numeric_limits::infinity())), kPlusMax, "+Inf should become +max"); + t.Equals(bitsOf(ps2_fclamp(-std::numeric_limits::infinity())), kMinusMax, "-Inf should become -max"); + t.Equals(bitsOf(ps2_fclamp(floatOf(0x7FC00000u))), kPlusMax, "positive NaN should become +max"); + t.Equals(bitsOf(ps2_fclamp(floatOf(0xFFC00000u))), kMinusMax, "negative NaN should become -max"); + t.Equals(bitsOf(ps2_fclamp(1.5f)), bitsOf(1.5f), "finite values should be unchanged"); + t.Equals(bitsOf(ps2_fclamp(floatOf(kPlusMax))), kPlusMax, "max should be unchanged"); + t.Equals(bitsOf(ps2_fclamp(-0.0f)), bitsOf(-0.0f), "negative zero should be unchanged"); + }); + + tc.Run("FPU ADD/SUB/MUL saturate instead of overflowing to Inf", [](TestCase &t) + { + const float maxf = floatOf(kPlusMax); + const float minf = floatOf(kMinusMax); + t.Equals(bitsOf(FPU_ADD_S(maxf, maxf)), kPlusMax, "max + max should saturate to +max"); + t.Equals(bitsOf(FPU_ADD_S(minf, minf)), kMinusMax, "-max + -max should saturate to -max"); + t.Equals(bitsOf(FPU_SUB_S(maxf, minf)), kPlusMax, "max - (-max) should saturate to +max"); + t.Equals(bitsOf(FPU_MUL_S(maxf, 2.0f)), kPlusMax, "max * 2 should saturate to +max"); + t.Equals(bitsOf(FPU_MUL_S(maxf, -2.0f)), kMinusMax, "max * -2 should saturate to -max"); + t.Equals(bitsOf(FPU_ADD_S(1.0f, 2.0f)), bitsOf(3.0f), "ordinary addition should be unchanged"); + t.Equals(bitsOf(FPU_MUL_S(1.5f, 4.0f)), bitsOf(6.0f), "ordinary multiplication should be unchanged"); + }); + + tc.Run("FPU ADD/SUB/MUL never produce NaN from Inf or NaN operands", [](TestCase &t) + { + const float inf = std::numeric_limits::infinity(); + const float nanf = floatOf(0x7FC00000u); + t.IsFalse(std::isnan(FPU_SUB_S(inf, inf)), "Inf - Inf should not be NaN"); + t.IsFalse(std::isnan(FPU_MUL_S(inf, 0.0f)), "Inf * 0 should not be NaN"); + t.Equals(bitsOf(FPU_MUL_S(inf, 0.0f)), bitsOf(0.0f), "Inf * 0 should be 0"); + t.IsFalse(std::isnan(FPU_ADD_S(nanf, 1.0f)), "NaN input should not produce NaN"); + t.IsFalse(std::isinf(FPU_ADD_S(nanf, 1.0f)), "NaN input should not leave an Inf"); + }); + + tc.Run("FPU DIV.S by zero gives +-max with the sign of fs xor ft", [](TestCase &t) + { + t.Equals(bitsOf(FPU_DIV_S(1.0f, 0.0f)), kPlusMax, "1/+0 should be +max"); + t.Equals(bitsOf(FPU_DIV_S(-1.0f, 0.0f)), kMinusMax, "-1/+0 should be -max"); + t.Equals(bitsOf(FPU_DIV_S(1.0f, -0.0f)), kMinusMax, "1/-0 should be -max"); + t.Equals(bitsOf(FPU_DIV_S(-1.0f, -0.0f)), kPlusMax, "-1/-0 should be +max"); + t.Equals(bitsOf(FPU_DIV_S(0.0f, 0.0f)), kPlusMax, "0/0 should be +max, not NaN"); + t.Equals(bitsOf(FPU_DIV_S(6.0f, 3.0f)), bitsOf(2.0f), "ordinary division should be unchanged"); + t.Equals(bitsOf(FPU_DIV_S(floatOf(kPlusMax), 0.5f)), kPlusMax, "overflowing quotient should saturate"); + }); + + tc.Run("FPU RSQRT.S computes fs / sqrt(ft)", [](TestCase &t) + { + t.Equals(bitsOf(FPU_RSQRT_S(4.0f, 16.0f)), bitsOf(1.0f), "4 / sqrt(16) should be 1"); + t.Equals(bitsOf(FPU_RSQRT_S(1.0f, 4.0f)), bitsOf(0.5f), "1 / sqrt(4) should be 0.5"); + t.Equals(bitsOf(FPU_RSQRT_S(1.0f, -4.0f)), bitsOf(0.5f), "the sign of the radicand should be ignored"); + t.Equals(bitsOf(FPU_RSQRT_S(2.0f, 0.0f)), kPlusMax, "x / sqrt(0) should be +max"); + t.Equals(bitsOf(FPU_RSQRT_S(-2.0f, 0.0f)), kMinusMax, "-x / sqrt(0) should be -max"); + }); + + tc.Run("FPU SQRT.S takes the square root of the magnitude", [](TestCase &t) + { + t.Equals(bitsOf(FPU_SQRT_S(16.0f)), bitsOf(4.0f), "sqrt(16) should be 4"); + t.Equals(bitsOf(FPU_SQRT_S(-16.0f)), bitsOf(4.0f), "sqrt(-16) should be 4, not NaN"); + }); + + tc.Run("VU vector ADD/SUB/MUL clamp every lane", [](TestCase &t) + { + // _mm_set_ps lists lanes 3..0 + const __m128 a = _mm_set_ps(1.0f, std::numeric_limits::infinity(), floatOf(kMinusMax), floatOf(kPlusMax)); + const __m128 b = _mm_set_ps(2.0f, 1.0f, floatOf(kMinusMax), floatOf(kPlusMax)); + + const __m128 sum = PS2_VADD(a, b); + t.Equals(laneBits(sum, 0), kPlusMax, "lane 0: max + max should saturate to +max"); + t.Equals(laneBits(sum, 1), kMinusMax, "lane 1: -max + -max should saturate to -max"); + t.Equals(laneBits(sum, 2), kPlusMax, "lane 2: +Inf + 1 should be +max"); + t.Equals(laneBits(sum, 3), bitsOf(3.0f), "lane 3: 1 + 2 should be 3"); + + const __m128 diff = PS2_VSUB(a, b); + t.Equals(laneBits(diff, 0), bitsOf(0.0f), "lane 0: max - max should be 0"); + t.Equals(laneBits(diff, 3), bitsOf(-1.0f), "lane 3: 1 - 2 should be -1"); + + const __m128 prod = PS2_VMUL(a, b); + t.Equals(laneBits(prod, 0), kPlusMax, "lane 0: max * max should saturate to +max"); + t.Equals(laneBits(prod, 1), kPlusMax, "lane 1: -max * -max should saturate to +max"); + t.Equals(laneBits(prod, 2), kPlusMax, "lane 2: +Inf * 1 should be +max"); + t.Equals(laneBits(prod, 3), bitsOf(2.0f), "lane 3: 1 * 2 should be 2"); + }); + + tc.Run("DIV.S and RSQRT.S translate to the PS2 semantics helpers", [](TestCase &t) + { + CodeGenerator gen({}, {}); + + // fd = f2, fs = f3, ft = f4 (all distinct so a wrong operand shows up) + const std::string div = gen.translateInstruction(makeCop1S(COP1_S_DIV, 2, 3, 4)); + t.IsTrue(div.find("ctx->f[2] = FPU_DIV_S(ctx->f[3], ctx->f[4])") != std::string::npos, + "DIV.S should write fs / ft to fd but got: " + div); + t.IsTrue(div.find("INFINITY") == std::string::npos, "DIV.S must not produce IEEE infinity but got: " + div); + + const std::string rsqrt = gen.translateInstruction(makeCop1S(COP1_S_RSQRT, 2, 3, 4)); + t.IsTrue(rsqrt.find("ctx->f[2] = FPU_RSQRT_S(ctx->f[3], ctx->f[4])") != std::string::npos, + "RSQRT.S should compute fs / sqrt(ft) but got: " + rsqrt); + }); + }); +}