From 2e91a2f6bddcf0b80d257725876f53c5a0e8b7f9 Mon Sep 17 00:00:00 2001 From: drako <98249188+drakolordx7@users.noreply.github.com> Date: Wed, 30 Sep 2026 18:19:13 -0500 Subject: [PATCH] Emulate PS2 float semantics for EE FPU results and DIV.S/RSQRT.S The EE FPU and the VUs have no Inf/NaN. A value with exponent 255 acts like a huge normal number, an overflowing result saturates to +-max (0x7F7FFFFF) and x/0 gives +-max with the sign of (fs xor ft). The generated code followed IEEE instead, so Inf/NaN leaked into later maths (a normalisation of a zero-length vector, for example). - ps2_runtime_macros.h: add ps2_fclamp/ps2_fdiv/ps2_frsqrt/ps2_vclamp. FPU_ADD_S/SUB_S/MUL_S clamp their inputs and result, FPU_DIV_S returns +-max on a zero divisor, FPU_SQRT_S takes sqrt(|x|), and PS2_VADD/VSUB/ VMUL clamp every lane. FPU_RSQRT_S is new. - fpu_translator.cpp: DIV.S no longer emits copysignf(INFINITY, ...) and wraps the DZ flag update and the division in one block. RSQRT.S was emitted as 1/sqrt(fs); the EE computes fs / sqrt(ft). Found while recompiling Killzone (SCUS-97402): RSQRT.S appears in 382 functions there (vector normalisation), and with the wrong operands the normals and lighting came out wrong. Verified with the new unit tests (edge values for each macro, and the translated DIV.S/RSQRT.S code with distinct fd/fs/ft), and by running Killzone with the change. Generated code must be regenerated and rebuilt to pick it up. CVT.W.S and the VU0 macro VDIV/VSQRT/VRSQRT Q-register operations are left to #215 and #198, which cover them. Made by drakolord and assisted with Claude Code. --- ps2xRecomp/src/lib/fpu_translator.cpp | 11 +- ps2xRuntime/include/ps2_runtime_macros.h | 67 +++++++-- ps2xTest/CMakeLists.txt | 1 + ps2xTest/src/main.cpp | 2 + ps2xTest/src/ps2_float_semantics_tests.cpp | 157 +++++++++++++++++++++ 5 files changed, 225 insertions(+), 13 deletions(-) create mode 100644 ps2xTest/src/ps2_float_semantics_tests.cpp diff --git a/ps2xRecomp/src/lib/fpu_translator.cpp b/ps2xRecomp/src/lib/fpu_translator.cpp index 944cee796..435db3aa9 100644 --- a/ps2xRecomp/src/lib/fpu_translator.cpp +++ b/ps2xRecomp/src/lib/fpu_translator.cpp @@ -53,10 +53,10 @@ namespace ps2recomp case COP1_S_MUL: return fmt::format("ctx->f[{}] = FPU_MUL_S(ctx->f[{}], ctx->f[{}]);", fd, fs, ft); case COP1_S_DIV: - return fmt::format("if (ctx->f[{}] == 0.0f) {{ ctx->fcr31 |= 0x100000; /* DZ flag */ " - "ctx->f[{}] = copysignf(INFINITY, ctx->f[{}] * 0.0f); }} " - "else ctx->f[{}] = ctx->f[{}] / ctx->f[{}];", - ft, fd, fs, fd, fs, ft); + // The EE has no Inf/NaN: x/0 gives +-max (sign of fs xor ft), see ps2_fdiv. + return fmt::format("{{ if (ctx->f[{}] == 0.0f) ctx->fcr31 |= 0x100000; /* DZ flag */ " + "ctx->f[{}] = FPU_DIV_S(ctx->f[{}], ctx->f[{}]); }}", + ft, fd, fs, ft); case COP1_S_SQRT: return fmt::format("ctx->f[{}] = FPU_SQRT_S(ctx->f[{}]);", fd, fs); case COP1_S_ABS: @@ -76,7 +76,8 @@ namespace ps2recomp case COP1_S_CVT_W: return fmt::format("{{ int32_t tmp = FPU_CVT_W_S(ctx->f[{}]); std::memcpy(&ctx->f[{}], &tmp, sizeof(tmp)); }}", fs, fd); case COP1_S_RSQRT: - return fmt::format("ctx->f[{}] = 1.0f / sqrtf(ctx->f[{}]);", fd, fs); + // RSQRT.S fd, fs, ft computes fs / sqrt(ft), not 1 / sqrt(fs). + return fmt::format("ctx->f[{}] = FPU_RSQRT_S(ctx->f[{}], ctx->f[{}]);", fd, fs, ft); case COP1_S_ADDA: return fmt::format("FPU_SET_ACC(ctx, FPU_ADD_S(ctx->f[{}], ctx->f[{}]));", fs, ft); case COP1_S_SUBA: diff --git a/ps2xRuntime/include/ps2_runtime_macros.h b/ps2xRuntime/include/ps2_runtime_macros.h index 0f19c9f24..cbcf9690e 100644 --- a/ps2xRuntime/include/ps2_runtime_macros.h +++ b/ps2xRuntime/include/ps2_runtime_macros.h @@ -140,10 +140,60 @@ static inline uint32_t ps2_plzcw32(uint32_t x) #define PS2_PXOR(a, b) _mm_xor_si128((__m128i)(a), (__m128i)(b)) #define PS2_PNOR(a, b) _mm_xor_si128(_mm_or_si128((__m128i)(a), (__m128i)(b)), _mm_set1_epi32(0xFFFFFFFF)) +// PS2 float semantics (EE FPU and VU). The PS2 has no Inf/NaN: a value with exponent 255 behaves like a huge +// normal number, an overflowing result saturates to +-max (0x7F7FFFFF) and x/0 gives +-max with the sign of +// (numerator xor denominator). Hosts that follow IEEE produce Inf/NaN here, which then poisons later maths +// (e.g. a normalisation of a zero-length vector). +inline float ps2_fclamp(float x) +{ + uint32_t b; + std::memcpy(&b, &x, sizeof(b)); + if ((b & 0x7F800000u) == 0x7F800000u) + b = (b & 0x80000000u) | 0x7F7FFFFFu; + std::memcpy(&x, &b, sizeof(b)); + return x; +} + +// Sign is sign(a) xor sign(b); used for a division by zero. +inline float ps2_fmax_signed(float a, float b) +{ + uint32_t ua, ub; + std::memcpy(&ua, &a, sizeof(ua)); + std::memcpy(&ub, &b, sizeof(ub)); + const uint32_t r = ((ua ^ ub) & 0x80000000u) | 0x7F7FFFFFu; + float f; + std::memcpy(&f, &r, sizeof(f)); + return f; +} + +inline float ps2_fdiv(float a, float b) +{ + a = ps2_fclamp(a); + b = ps2_fclamp(b); + return (b == 0.0f) ? ps2_fmax_signed(a, b) : ps2_fclamp(a / b); +} + +// fs / sqrt(|ft|); the radicand's sign is ignored. +inline float ps2_frsqrt(float a, float b) +{ + a = ps2_fclamp(a); + b = ps2_fclamp(b); + return (b == 0.0f) ? ps2_fmax_signed(a, b) : ps2_fclamp(a / std::sqrt(std::fabs(b))); +} + +inline __m128 ps2_vclamp(__m128 v) +{ + const __m128i bits = _mm_castps_si128(v); + const __m128i expMask = _mm_set1_epi32(0x7F800000); + const __m128i special = _mm_cmpeq_epi32(_mm_and_si128(bits, expMask), expMask); + const __m128i saturated = _mm_or_si128(_mm_and_si128(bits, _mm_set1_epi32((int)0x80000000)), _mm_set1_epi32(0x7F7FFFFF)); + return _mm_castsi128_ps(_mm_or_si128(_mm_and_si128(special, saturated), _mm_andnot_si128(special, bits))); +} + // PS2 VU (Vector Unit) operations -#define PS2_VADD(a, b) _mm_add_ps((__m128)(a), (__m128)(b)) -#define PS2_VSUB(a, b) _mm_sub_ps((__m128)(a), (__m128)(b)) -#define PS2_VMUL(a, b) _mm_mul_ps((__m128)(a), (__m128)(b)) +#define PS2_VADD(a, b) ps2_vclamp(_mm_add_ps(ps2_vclamp((__m128)(a)), ps2_vclamp((__m128)(b)))) +#define PS2_VSUB(a, b) ps2_vclamp(_mm_sub_ps(ps2_vclamp((__m128)(a)), ps2_vclamp((__m128)(b)))) +#define PS2_VMUL(a, b) ps2_vclamp(_mm_mul_ps(ps2_vclamp((__m128)(a)), ps2_vclamp((__m128)(b)))) #define PS2_VDIV(a, b) _mm_div_ps((__m128)(a), (__m128)(b)) #define PS2_VMULQ(a, q) _mm_mul_ps((__m128)(a), _mm_set1_ps(q)) #define PS2_VBLEND(a, b, mask) PS2_BLENDV_PS((__m128)(a), (__m128)(b), (__m128)(mask)) @@ -605,11 +655,12 @@ inline __m128i ps2_u64_to_epi64_pair(uint64_t value) // FPU (COP1) operations #define FPU_SET_ACC(ctx, res) (ctx->f_acc = res) -#define FPU_ADD_S(a, b) ((float)(a) + (float)(b)) -#define FPU_SUB_S(a, b) ((float)(a) - (float)(b)) -#define FPU_MUL_S(a, b) ((float)(a) * (float)(b)) -#define FPU_DIV_S(a, b) ((float)(a) / (float)(b)) -#define FPU_SQRT_S(a) sqrtf((float)(a)) +#define FPU_ADD_S(a, b) ps2_fclamp(ps2_fclamp((float)(a)) + ps2_fclamp((float)(b))) +#define FPU_SUB_S(a, b) ps2_fclamp(ps2_fclamp((float)(a)) - ps2_fclamp((float)(b))) +#define FPU_MUL_S(a, b) ps2_fclamp(ps2_fclamp((float)(a)) * ps2_fclamp((float)(b))) +#define FPU_DIV_S(a, b) ps2_fdiv((float)(a), (float)(b)) +#define FPU_RSQRT_S(a, b) ps2_frsqrt((float)(a), (float)(b)) +#define FPU_SQRT_S(a) std::sqrt(std::fabs(ps2_fclamp((float)(a)))) #define FPU_ABS_S(a) fabsf((float)(a)) #define FPU_MOV_S(a) ((float)(a)) #define FPU_NEG_S(a) (-(float)(a)) diff --git a/ps2xTest/CMakeLists.txt b/ps2xTest/CMakeLists.txt index 5db04b923..3eaa8b3d2 100644 --- a/ps2xTest/CMakeLists.txt +++ b/ps2xTest/CMakeLists.txt @@ -37,6 +37,7 @@ add_library(ps2_test_lib STATIC src/ps2_sif_dma_tests.cpp src/ps2_recompiler_tests.cpp src/ps2_runtime_expansion_tests.cpp + src/ps2_float_semantics_tests.cpp ) target_include_directories(ps2_test_lib PRIVATE diff --git a/ps2xTest/src/main.cpp b/ps2xTest/src/main.cpp index 40a68b16b..7b7889148 100644 --- a/ps2xTest/src/main.cpp +++ b/ps2xTest/src/main.cpp @@ -18,6 +18,7 @@ void register_ps2_sif_rpc_tests(); void register_ps2_sif_dma_tests(); void register_ps2_recompiler_tests(); void register_ps2_runtime_expansion_tests(); +void register_ps2_float_semantics_tests(); void reset_ps2_test_function_table(); int main() @@ -40,6 +41,7 @@ int main() register_ps2_sif_dma_tests(); register_ps2_recompiler_tests(); register_ps2_runtime_expansion_tests(); + register_ps2_float_semantics_tests(); int res = MiniTest::Run(); std::cout.flush(); std::cerr.flush(); diff --git a/ps2xTest/src/ps2_float_semantics_tests.cpp b/ps2xTest/src/ps2_float_semantics_tests.cpp new file mode 100644 index 000000000..adfec5852 --- /dev/null +++ b/ps2xTest/src/ps2_float_semantics_tests.cpp @@ -0,0 +1,157 @@ +#include "MiniTest.h" +#include "ps2recomp/code_generator.h" +#include "ps2recomp/instructions.h" +#include "ps2recomp/types.h" +#include "ps2_runtime_macros.h" + +#include +#include +#include +#include +#include + +using namespace ps2recomp; + +namespace +{ + constexpr uint32_t kPlusMax = 0x7F7FFFFFu; + constexpr uint32_t kMinusMax = 0xFF7FFFFFu; + + uint32_t bitsOf(float f) + { + uint32_t b; + std::memcpy(&b, &f, sizeof(b)); + return b; + } + + float floatOf(uint32_t b) + { + float f; + std::memcpy(&f, &b, sizeof(f)); + return f; + } + + Instruction makeCop1S(uint32_t function, uint8_t fd, uint8_t fs, uint8_t ft) + { + Instruction inst{}; + inst.opcode = OPCODE_COP1; + inst.rs = COP1_S; + inst.rt = ft; + inst.rd = fs; + inst.sa = fd; + inst.function = function; + return inst; + } + + uint32_t laneBits(__m128 v, int lane) + { + float lanes[4]; + _mm_storeu_ps(lanes, v); + return bitsOf(lanes[lane]); + } +} + +void register_ps2_float_semantics_tests() +{ + MiniTest::Case("PS2FloatSemantics", [](TestCase &tc) + { + tc.Run("ps2_fclamp turns Inf and NaN into +-max and leaves finite values alone", [](TestCase &t) + { + t.Equals(bitsOf(ps2_fclamp(std::numeric_limits::infinity())), kPlusMax, "+Inf should become +max"); + t.Equals(bitsOf(ps2_fclamp(-std::numeric_limits::infinity())), kMinusMax, "-Inf should become -max"); + t.Equals(bitsOf(ps2_fclamp(floatOf(0x7FC00000u))), kPlusMax, "positive NaN should become +max"); + t.Equals(bitsOf(ps2_fclamp(floatOf(0xFFC00000u))), kMinusMax, "negative NaN should become -max"); + t.Equals(bitsOf(ps2_fclamp(1.5f)), bitsOf(1.5f), "finite values should be unchanged"); + t.Equals(bitsOf(ps2_fclamp(floatOf(kPlusMax))), kPlusMax, "max should be unchanged"); + t.Equals(bitsOf(ps2_fclamp(-0.0f)), bitsOf(-0.0f), "negative zero should be unchanged"); + }); + + tc.Run("FPU ADD/SUB/MUL saturate instead of overflowing to Inf", [](TestCase &t) + { + const float maxf = floatOf(kPlusMax); + const float minf = floatOf(kMinusMax); + t.Equals(bitsOf(FPU_ADD_S(maxf, maxf)), kPlusMax, "max + max should saturate to +max"); + t.Equals(bitsOf(FPU_ADD_S(minf, minf)), kMinusMax, "-max + -max should saturate to -max"); + t.Equals(bitsOf(FPU_SUB_S(maxf, minf)), kPlusMax, "max - (-max) should saturate to +max"); + t.Equals(bitsOf(FPU_MUL_S(maxf, 2.0f)), kPlusMax, "max * 2 should saturate to +max"); + t.Equals(bitsOf(FPU_MUL_S(maxf, -2.0f)), kMinusMax, "max * -2 should saturate to -max"); + t.Equals(bitsOf(FPU_ADD_S(1.0f, 2.0f)), bitsOf(3.0f), "ordinary addition should be unchanged"); + t.Equals(bitsOf(FPU_MUL_S(1.5f, 4.0f)), bitsOf(6.0f), "ordinary multiplication should be unchanged"); + }); + + tc.Run("FPU ADD/SUB/MUL never produce NaN from Inf or NaN operands", [](TestCase &t) + { + const float inf = std::numeric_limits::infinity(); + const float nanf = floatOf(0x7FC00000u); + t.IsFalse(std::isnan(FPU_SUB_S(inf, inf)), "Inf - Inf should not be NaN"); + t.IsFalse(std::isnan(FPU_MUL_S(inf, 0.0f)), "Inf * 0 should not be NaN"); + t.Equals(bitsOf(FPU_MUL_S(inf, 0.0f)), bitsOf(0.0f), "Inf * 0 should be 0"); + t.IsFalse(std::isnan(FPU_ADD_S(nanf, 1.0f)), "NaN input should not produce NaN"); + t.IsFalse(std::isinf(FPU_ADD_S(nanf, 1.0f)), "NaN input should not leave an Inf"); + }); + + tc.Run("FPU DIV.S by zero gives +-max with the sign of fs xor ft", [](TestCase &t) + { + t.Equals(bitsOf(FPU_DIV_S(1.0f, 0.0f)), kPlusMax, "1/+0 should be +max"); + t.Equals(bitsOf(FPU_DIV_S(-1.0f, 0.0f)), kMinusMax, "-1/+0 should be -max"); + t.Equals(bitsOf(FPU_DIV_S(1.0f, -0.0f)), kMinusMax, "1/-0 should be -max"); + t.Equals(bitsOf(FPU_DIV_S(-1.0f, -0.0f)), kPlusMax, "-1/-0 should be +max"); + t.Equals(bitsOf(FPU_DIV_S(0.0f, 0.0f)), kPlusMax, "0/0 should be +max, not NaN"); + t.Equals(bitsOf(FPU_DIV_S(6.0f, 3.0f)), bitsOf(2.0f), "ordinary division should be unchanged"); + t.Equals(bitsOf(FPU_DIV_S(floatOf(kPlusMax), 0.5f)), kPlusMax, "overflowing quotient should saturate"); + }); + + tc.Run("FPU RSQRT.S computes fs / sqrt(ft)", [](TestCase &t) + { + t.Equals(bitsOf(FPU_RSQRT_S(4.0f, 16.0f)), bitsOf(1.0f), "4 / sqrt(16) should be 1"); + t.Equals(bitsOf(FPU_RSQRT_S(1.0f, 4.0f)), bitsOf(0.5f), "1 / sqrt(4) should be 0.5"); + t.Equals(bitsOf(FPU_RSQRT_S(1.0f, -4.0f)), bitsOf(0.5f), "the sign of the radicand should be ignored"); + t.Equals(bitsOf(FPU_RSQRT_S(2.0f, 0.0f)), kPlusMax, "x / sqrt(0) should be +max"); + t.Equals(bitsOf(FPU_RSQRT_S(-2.0f, 0.0f)), kMinusMax, "-x / sqrt(0) should be -max"); + }); + + tc.Run("FPU SQRT.S takes the square root of the magnitude", [](TestCase &t) + { + t.Equals(bitsOf(FPU_SQRT_S(16.0f)), bitsOf(4.0f), "sqrt(16) should be 4"); + t.Equals(bitsOf(FPU_SQRT_S(-16.0f)), bitsOf(4.0f), "sqrt(-16) should be 4, not NaN"); + }); + + tc.Run("VU vector ADD/SUB/MUL clamp every lane", [](TestCase &t) + { + // _mm_set_ps lists lanes 3..0 + const __m128 a = _mm_set_ps(1.0f, std::numeric_limits::infinity(), floatOf(kMinusMax), floatOf(kPlusMax)); + const __m128 b = _mm_set_ps(2.0f, 1.0f, floatOf(kMinusMax), floatOf(kPlusMax)); + + const __m128 sum = PS2_VADD(a, b); + t.Equals(laneBits(sum, 0), kPlusMax, "lane 0: max + max should saturate to +max"); + t.Equals(laneBits(sum, 1), kMinusMax, "lane 1: -max + -max should saturate to -max"); + t.Equals(laneBits(sum, 2), kPlusMax, "lane 2: +Inf + 1 should be +max"); + t.Equals(laneBits(sum, 3), bitsOf(3.0f), "lane 3: 1 + 2 should be 3"); + + const __m128 diff = PS2_VSUB(a, b); + t.Equals(laneBits(diff, 0), bitsOf(0.0f), "lane 0: max - max should be 0"); + t.Equals(laneBits(diff, 3), bitsOf(-1.0f), "lane 3: 1 - 2 should be -1"); + + const __m128 prod = PS2_VMUL(a, b); + t.Equals(laneBits(prod, 0), kPlusMax, "lane 0: max * max should saturate to +max"); + t.Equals(laneBits(prod, 1), kPlusMax, "lane 1: -max * -max should saturate to +max"); + t.Equals(laneBits(prod, 2), kPlusMax, "lane 2: +Inf * 1 should be +max"); + t.Equals(laneBits(prod, 3), bitsOf(2.0f), "lane 3: 1 * 2 should be 2"); + }); + + tc.Run("DIV.S and RSQRT.S translate to the PS2 semantics helpers", [](TestCase &t) + { + CodeGenerator gen({}, {}); + + // fd = f2, fs = f3, ft = f4 (all distinct so a wrong operand shows up) + const std::string div = gen.translateInstruction(makeCop1S(COP1_S_DIV, 2, 3, 4)); + t.IsTrue(div.find("ctx->f[2] = FPU_DIV_S(ctx->f[3], ctx->f[4])") != std::string::npos, + "DIV.S should write fs / ft to fd but got: " + div); + t.IsTrue(div.find("INFINITY") == std::string::npos, "DIV.S must not produce IEEE infinity but got: " + div); + + const std::string rsqrt = gen.translateInstruction(makeCop1S(COP1_S_RSQRT, 2, 3, 4)); + t.IsTrue(rsqrt.find("ctx->f[2] = FPU_RSQRT_S(ctx->f[3], ctx->f[4])") != std::string::npos, + "RSQRT.S should compute fs / sqrt(ft) but got: " + rsqrt); + }); + }); +}