From d8a853795f1b0bbae854981e8d9de78952b373dd Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Branimir=20Karad=C5=BEi=C4=87?= Date: Thu, 3 Sep 2026 01:41:39 +0000 Subject: [PATCH] Fixed build. (#422) --- include/bx/inline/simd256_ref.inl | 244 +++++++++++++++--------------- include/bx/simd_t.h | 6 + tests/math_test.cpp | 2 +- tests/scanner_test.cpp | 1 + 4 files changed, 130 insertions(+), 123 deletions(-) diff --git a/include/bx/inline/simd256_ref.inl b/include/bx/inline/simd256_ref.inl index 8ca60e7..8a79603 100644 --- a/include/bx/inline/simd256_ref.inl +++ b/include/bx/inline/simd256_ref.inl @@ -11,7 +11,7 @@ namespace bx { // 256-bit reference delegates to two 128-bit operations. -#if BX_SIMD_LANGEXT +#if BX_SIMD256_LANGEXT typedef float simd256_f32_langext_t __attribute__((__vector_size__(32), __aligned__(32))); typedef double simd256_f64_langext_t __attribute__((__vector_size__(32), __aligned__(32))); typedef int8_t simd256_i8_langext_t __attribute__((__vector_size__(32), __aligned__(32))); @@ -22,7 +22,7 @@ namespace bx typedef uint16_t simd256_u16_langext_t __attribute__((__vector_size__(32), __aligned__(32))); typedef uint32_t simd256_u32_langext_t __attribute__((__vector_size__(32), __aligned__(32))); typedef uint64_t simd256_u64_langext_t __attribute__((__vector_size__(32), __aligned__(32))); -#endif // BX_SIMD_LANGEXT +#endif // BX_SIMD256_LANGEXT #if !BX_SIMD_AVX @@ -151,7 +151,7 @@ namespace bx template<> inline BX_CONST_FUNC simd256_ref_t simd256_i32_itof(simd256_ref_t _a) { -#if BX_SIMD_LANGEXT +#if BX_SIMD256_LANGEXT const simd256_i32_langext_t a = bitCast(_a); const simd256_f32_langext_t converted = __builtin_convertvector(a, simd256_f32_langext_t); const simd256_ref_t result = bitCast(converted); @@ -161,13 +161,13 @@ namespace bx result.lo = simd128_i32_itof(_a.lo); result.hi = simd128_i32_itof(_a.hi); return result; -#endif // BX_SIMD_LANGEXT +#endif // BX_SIMD256_LANGEXT } template<> inline BX_CONST_FUNC simd256_ref_t simd256_f32_ftoi_trunc(simd256_ref_t _a) { -#if BX_SIMD_LANGEXT +#if BX_SIMD256_LANGEXT const simd256_f32_langext_t a = bitCast(_a); const simd256_i32_langext_t converted = __builtin_convertvector(a, simd256_i32_langext_t); const simd256_ref_t result = bitCast(converted); @@ -177,7 +177,7 @@ namespace bx result.lo = simd128_f32_ftoi_trunc(_a.lo); result.hi = simd128_f32_ftoi_trunc(_a.hi); return result; -#endif // BX_SIMD_LANGEXT +#endif // BX_SIMD256_LANGEXT } template<> @@ -192,7 +192,7 @@ namespace bx template<> inline BX_CONSTEXPR_FUNC simd256_ref_t simd256_f32_add(simd256_ref_t _a, simd256_ref_t _b) { -#if BX_SIMD_LANGEXT +#if BX_SIMD256_LANGEXT const simd256_f32_langext_t a = bitCast(_a); const simd256_f32_langext_t b = bitCast(_b); const simd256_f32_langext_t sum = a + b; @@ -203,13 +203,13 @@ namespace bx result.lo = simd128_f32_add(_a.lo, _b.lo); result.hi = simd128_f32_add(_a.hi, _b.hi); return result; -#endif // BX_SIMD_LANGEXT +#endif // BX_SIMD256_LANGEXT } template<> inline BX_CONSTEXPR_FUNC simd256_ref_t simd256_f32_sub(simd256_ref_t _a, simd256_ref_t _b) { -#if BX_SIMD_LANGEXT +#if BX_SIMD256_LANGEXT const simd256_f32_langext_t a = bitCast(_a); const simd256_f32_langext_t b = bitCast(_b); const simd256_f32_langext_t diff = a - b; @@ -220,13 +220,13 @@ namespace bx result.lo = simd128_f32_sub(_a.lo, _b.lo); result.hi = simd128_f32_sub(_a.hi, _b.hi); return result; -#endif // BX_SIMD_LANGEXT +#endif // BX_SIMD256_LANGEXT } template<> inline BX_CONSTEXPR_FUNC simd256_ref_t simd256_f32_mul(simd256_ref_t _a, simd256_ref_t _b) { -#if BX_SIMD_LANGEXT +#if BX_SIMD256_LANGEXT const simd256_f32_langext_t a = bitCast(_a); const simd256_f32_langext_t b = bitCast(_b); const simd256_f32_langext_t prod = a * b; @@ -237,13 +237,13 @@ namespace bx result.lo = simd128_f32_mul(_a.lo, _b.lo); result.hi = simd128_f32_mul(_a.hi, _b.hi); return result; -#endif // BX_SIMD_LANGEXT +#endif // BX_SIMD256_LANGEXT } template<> inline BX_CONSTEXPR_FUNC simd256_ref_t simd256_f32_div(simd256_ref_t _a, simd256_ref_t _b) { -#if BX_SIMD_LANGEXT +#if BX_SIMD256_LANGEXT const simd256_f32_langext_t a = bitCast(_a); const simd256_f32_langext_t b = bitCast(_b); const simd256_f32_langext_t quot = a / b; @@ -254,7 +254,7 @@ namespace bx result.lo = simd128_f32_div(_a.lo, _b.lo); result.hi = simd128_f32_div(_a.hi, _b.hi); return result; -#endif // BX_SIMD_LANGEXT +#endif // BX_SIMD256_LANGEXT } template<> @@ -287,14 +287,14 @@ namespace bx template<> inline BX_CONST_FUNC simd256_ref_t simd256_f32_neg(simd256_ref_t _a) { -#if BX_SIMD_LANGEXT +#if BX_SIMD256_LANGEXT const simd256_f32_langext_t a = bitCast(_a); const simd256_f32_langext_t neg = -a; const simd256_ref_t result = bitCast(neg); return result; #else return simd_f32_neg_ni(_a); -#endif // BX_SIMD_LANGEXT +#endif // BX_SIMD256_LANGEXT } template<> @@ -332,7 +332,7 @@ namespace bx template<> inline BX_CONST_FUNC simd256_ref_t simd256_f32_lerp(simd256_ref_t _a, simd256_ref_t _b, simd256_ref_t _s) { -#if BX_SIMD_LANGEXT +#if BX_SIMD256_LANGEXT const simd256_f32_langext_t a = bitCast(_a); const simd256_f32_langext_t b = bitCast(_b); const simd256_f32_langext_t s = bitCast(_s); @@ -343,7 +343,7 @@ namespace bx return result; #else return simd_f32_lerp_ni(_a, _b, _s); -#endif // BX_SIMD_LANGEXT +#endif // BX_SIMD256_LANGEXT } template<> @@ -355,7 +355,7 @@ namespace bx template<> inline BX_CONSTEXPR_FUNC simd256_ref_t simd256_f32_rcp_est(simd256_ref_t _a) { -#if BX_SIMD_LANGEXT +#if BX_SIMD256_LANGEXT const simd256_f32_langext_t one = {1.0f, 1.0f, 1.0f, 1.0f, 1.0f, 1.0f, 1.0f, 1.0f}; const simd256_f32_langext_t a = bitCast(_a); const simd256_f32_langext_t quot = one / a; @@ -366,7 +366,7 @@ namespace bx result.lo = simd128_f32_rcp_est(_a.lo); result.hi = simd128_f32_rcp_est(_a.hi); return result; -#endif // BX_SIMD_LANGEXT +#endif // BX_SIMD256_LANGEXT } template<> @@ -420,7 +420,7 @@ namespace bx template<> inline BX_CONSTEXPR_FUNC simd256_ref_t simd256_f32_cmpeq(simd256_ref_t _a, simd256_ref_t _b) { -#if BX_SIMD_LANGEXT +#if BX_SIMD256_LANGEXT const simd256_f32_langext_t a = bitCast(_a); const simd256_f32_langext_t b = bitCast(_b); const simd256_f32_langext_t cmp = a == b; @@ -431,13 +431,13 @@ namespace bx result.lo = simd128_f32_cmpeq(_a.lo, _b.lo); result.hi = simd128_f32_cmpeq(_a.hi, _b.hi); return result; -#endif // BX_SIMD_LANGEXT +#endif // BX_SIMD256_LANGEXT } template<> inline BX_CONST_FUNC simd256_ref_t simd256_f32_cmpneq(simd256_ref_t _a, simd256_ref_t _b) { -#if BX_SIMD_LANGEXT +#if BX_SIMD256_LANGEXT const simd256_f32_langext_t a = bitCast(_a); const simd256_f32_langext_t b = bitCast(_b); const simd256_f32_langext_t cmp = a != b; @@ -445,13 +445,13 @@ namespace bx return result; #else return simd_f32_cmpneq_ni(_a, _b); -#endif // BX_SIMD_LANGEXT +#endif // BX_SIMD256_LANGEXT } template<> inline BX_CONSTEXPR_FUNC simd256_ref_t simd256_f32_cmplt(simd256_ref_t _a, simd256_ref_t _b) { -#if BX_SIMD_LANGEXT +#if BX_SIMD256_LANGEXT const simd256_f32_langext_t a = bitCast(_a); const simd256_f32_langext_t b = bitCast(_b); const simd256_f32_langext_t cmp = a < b; @@ -462,13 +462,13 @@ namespace bx result.lo = simd128_f32_cmplt(_a.lo, _b.lo); result.hi = simd128_f32_cmplt(_a.hi, _b.hi); return result; -#endif // BX_SIMD_LANGEXT +#endif // BX_SIMD256_LANGEXT } template<> inline BX_CONSTEXPR_FUNC simd256_ref_t simd256_f32_cmpgt(simd256_ref_t _a, simd256_ref_t _b) { -#if BX_SIMD_LANGEXT +#if BX_SIMD256_LANGEXT const simd256_f32_langext_t a = bitCast(_a); const simd256_f32_langext_t b = bitCast(_b); const simd256_f32_langext_t cmp = a > b; @@ -479,13 +479,13 @@ namespace bx result.lo = simd128_f32_cmpgt(_a.lo, _b.lo); result.hi = simd128_f32_cmpgt(_a.hi, _b.hi); return result; -#endif // BX_SIMD_LANGEXT +#endif // BX_SIMD256_LANGEXT } template<> inline BX_CONSTEXPR_FUNC simd256_ref_t simd256_f32_cmple(simd256_ref_t _a, simd256_ref_t _b) { -#if BX_SIMD_LANGEXT +#if BX_SIMD256_LANGEXT const simd256_f32_langext_t a = bitCast(_a); const simd256_f32_langext_t b = bitCast(_b); const simd256_f32_langext_t cmp = a <= b; @@ -496,13 +496,13 @@ namespace bx result.lo = simd128_f32_cmple(_a.lo, _b.lo); result.hi = simd128_f32_cmple(_a.hi, _b.hi); return result; -#endif // BX_SIMD_LANGEXT +#endif // BX_SIMD256_LANGEXT } template<> inline BX_CONSTEXPR_FUNC simd256_ref_t simd256_f32_cmpge(simd256_ref_t _a, simd256_ref_t _b) { -#if BX_SIMD_LANGEXT +#if BX_SIMD256_LANGEXT const simd256_f32_langext_t a = bitCast(_a); const simd256_f32_langext_t b = bitCast(_b); const simd256_f32_langext_t cmp = a >= b; @@ -513,13 +513,13 @@ namespace bx result.lo = simd128_f32_cmpge(_a.lo, _b.lo); result.hi = simd128_f32_cmpge(_a.hi, _b.hi); return result; -#endif // BX_SIMD_LANGEXT +#endif // BX_SIMD256_LANGEXT } template<> inline BX_CONSTEXPR_FUNC simd256_ref_t simd256_i32_add(simd256_ref_t _a, simd256_ref_t _b) { -#if BX_SIMD_LANGEXT +#if BX_SIMD256_LANGEXT const simd256_i32_langext_t a = bitCast(_a); const simd256_i32_langext_t b = bitCast(_b); const simd256_i32_langext_t sum = a + b; @@ -530,13 +530,13 @@ namespace bx result.lo = simd128_i32_add(_a.lo, _b.lo); result.hi = simd128_i32_add(_a.hi, _b.hi); return result; -#endif // BX_SIMD_LANGEXT +#endif // BX_SIMD256_LANGEXT } template<> inline BX_CONSTEXPR_FUNC simd256_ref_t simd256_i32_sub(simd256_ref_t _a, simd256_ref_t _b) { -#if BX_SIMD_LANGEXT +#if BX_SIMD256_LANGEXT const simd256_i32_langext_t a = bitCast(_a); const simd256_i32_langext_t b = bitCast(_b); const simd256_i32_langext_t diff = a - b; @@ -547,20 +547,20 @@ namespace bx result.lo = simd128_i32_sub(_a.lo, _b.lo); result.hi = simd128_i32_sub(_a.hi, _b.hi); return result; -#endif // BX_SIMD_LANGEXT +#endif // BX_SIMD256_LANGEXT } template<> inline BX_CONST_FUNC simd256_ref_t simd256_i32_neg(simd256_ref_t _a) { -#if BX_SIMD_LANGEXT +#if BX_SIMD256_LANGEXT const simd256_i32_langext_t a = bitCast(_a); const simd256_i32_langext_t neg = -a; const simd256_ref_t result = bitCast(neg); return result; #else return simd_i32_neg_ni(_a); -#endif // BX_SIMD_LANGEXT +#endif // BX_SIMD256_LANGEXT } template<> @@ -598,7 +598,7 @@ namespace bx template<> inline BX_CONSTEXPR_FUNC simd256_ref_t simd256_i32_cmpeq(simd256_ref_t _a, simd256_ref_t _b) { -#if BX_SIMD_LANGEXT +#if BX_SIMD256_LANGEXT const simd256_i32_langext_t a = bitCast(_a); const simd256_i32_langext_t b = bitCast(_b); const simd256_i32_langext_t cmp = a == b; @@ -609,13 +609,13 @@ namespace bx result.lo = simd128_i32_cmpeq(_a.lo, _b.lo); result.hi = simd128_i32_cmpeq(_a.hi, _b.hi); return result; -#endif // BX_SIMD_LANGEXT +#endif // BX_SIMD256_LANGEXT } template<> inline BX_CONSTEXPR_FUNC simd256_ref_t simd256_i32_cmplt(simd256_ref_t _a, simd256_ref_t _b) { -#if BX_SIMD_LANGEXT +#if BX_SIMD256_LANGEXT const simd256_i32_langext_t a = bitCast(_a); const simd256_i32_langext_t b = bitCast(_b); const simd256_i32_langext_t cmp = a < b; @@ -626,13 +626,13 @@ namespace bx result.lo = simd128_i32_cmplt(_a.lo, _b.lo); result.hi = simd128_i32_cmplt(_a.hi, _b.hi); return result; -#endif // BX_SIMD_LANGEXT +#endif // BX_SIMD256_LANGEXT } template<> inline BX_CONSTEXPR_FUNC simd256_ref_t simd256_i32_cmpgt(simd256_ref_t _a, simd256_ref_t _b) { -#if BX_SIMD_LANGEXT +#if BX_SIMD256_LANGEXT const simd256_i32_langext_t a = bitCast(_a); const simd256_i32_langext_t b = bitCast(_b); const simd256_i32_langext_t cmp = a > b; @@ -643,13 +643,13 @@ namespace bx result.lo = simd128_i32_cmpgt(_a.lo, _b.lo); result.hi = simd128_i32_cmpgt(_a.hi, _b.hi); return result; -#endif // BX_SIMD_LANGEXT +#endif // BX_SIMD256_LANGEXT } template<> inline BX_CONSTEXPR_FUNC simd256_ref_t simd256_u32_add(simd256_ref_t _a, simd256_ref_t _b) { -#if BX_SIMD_LANGEXT +#if BX_SIMD256_LANGEXT const simd256_u32_langext_t a = bitCast(_a); const simd256_u32_langext_t b = bitCast(_b); const simd256_u32_langext_t sum = a + b; @@ -660,13 +660,13 @@ namespace bx result.lo = simd128_u32_add(_a.lo, _b.lo); result.hi = simd128_u32_add(_a.hi, _b.hi); return result; -#endif // BX_SIMD_LANGEXT +#endif // BX_SIMD256_LANGEXT } template<> inline BX_CONSTEXPR_FUNC simd256_ref_t simd256_u32_sub(simd256_ref_t _a, simd256_ref_t _b) { -#if BX_SIMD_LANGEXT +#if BX_SIMD256_LANGEXT const simd256_u32_langext_t a = bitCast(_a); const simd256_u32_langext_t b = bitCast(_b); const simd256_u32_langext_t diff = a - b; @@ -677,13 +677,13 @@ namespace bx result.lo = simd128_u32_sub(_a.lo, _b.lo); result.hi = simd128_u32_sub(_a.hi, _b.hi); return result; -#endif // BX_SIMD_LANGEXT +#endif // BX_SIMD256_LANGEXT } template<> inline BX_CONSTEXPR_FUNC simd256_ref_t simd256_u32_mul(simd256_ref_t _a, simd256_ref_t _b) { -#if BX_SIMD_LANGEXT +#if BX_SIMD256_LANGEXT const simd256_u32_langext_t a = bitCast(_a); const simd256_u32_langext_t b = bitCast(_b); const simd256_u32_langext_t prod = a * b; @@ -694,7 +694,7 @@ namespace bx result.lo = simd128_u32_mul(_a.lo, _b.lo); result.hi = simd128_u32_mul(_a.hi, _b.hi); return result; -#endif // BX_SIMD_LANGEXT +#endif // BX_SIMD256_LANGEXT } template<> @@ -726,7 +726,7 @@ namespace bx template<> inline BX_CONSTEXPR_FUNC simd256_ref_t simd256_u32_cmpeq(simd256_ref_t _a, simd256_ref_t _b) { -#if BX_SIMD_LANGEXT +#if BX_SIMD256_LANGEXT const simd256_u32_langext_t a = bitCast(_a); const simd256_u32_langext_t b = bitCast(_b); const simd256_u32_langext_t cmp = a == b; @@ -737,7 +737,7 @@ namespace bx result.lo = simd128_u32_cmpeq(_a.lo, _b.lo); result.hi = simd128_u32_cmpeq(_a.hi, _b.hi); return result; -#endif // BX_SIMD_LANGEXT +#endif // BX_SIMD256_LANGEXT } template<> @@ -779,7 +779,7 @@ namespace bx template<> inline BX_CONSTEXPR_FUNC simd256_ref_t simd256_u32_cmplt(simd256_ref_t _a, simd256_ref_t _b) { -#if BX_SIMD_LANGEXT +#if BX_SIMD256_LANGEXT const simd256_u32_langext_t a = bitCast(_a); const simd256_u32_langext_t b = bitCast(_b); const simd256_u32_langext_t cmp = a < b; @@ -790,13 +790,13 @@ namespace bx result.lo = simd128_u32_cmplt(_a.lo, _b.lo); result.hi = simd128_u32_cmplt(_a.hi, _b.hi); return result; -#endif // BX_SIMD_LANGEXT +#endif // BX_SIMD256_LANGEXT } template<> inline BX_CONSTEXPR_FUNC simd256_ref_t simd256_u32_cmpgt(simd256_ref_t _a, simd256_ref_t _b) { -#if BX_SIMD_LANGEXT +#if BX_SIMD256_LANGEXT const simd256_u32_langext_t a = bitCast(_a); const simd256_u32_langext_t b = bitCast(_b); const simd256_u32_langext_t cmp = a > b; @@ -807,13 +807,13 @@ namespace bx result.lo = simd128_u32_cmpgt(_a.lo, _b.lo); result.hi = simd128_u32_cmpgt(_a.hi, _b.hi); return result; -#endif // BX_SIMD_LANGEXT +#endif // BX_SIMD256_LANGEXT } template<> inline BX_CONSTEXPR_FUNC simd256_ref_t simd256_and(simd256_ref_t _a, simd256_ref_t _b) { -#if BX_SIMD_LANGEXT +#if BX_SIMD256_LANGEXT const simd256_u32_langext_t a = bitCast(_a); const simd256_u32_langext_t b = bitCast(_b); const simd256_u32_langext_t masked = a & b; @@ -824,13 +824,13 @@ namespace bx result.lo = simd128_and(_a.lo, _b.lo); result.hi = simd128_and(_a.hi, _b.hi); return result; -#endif // BX_SIMD_LANGEXT +#endif // BX_SIMD256_LANGEXT } template<> inline BX_CONSTEXPR_FUNC simd256_ref_t simd256_or(simd256_ref_t _a, simd256_ref_t _b) { -#if BX_SIMD_LANGEXT +#if BX_SIMD256_LANGEXT const simd256_u32_langext_t a = bitCast(_a); const simd256_u32_langext_t b = bitCast(_b); const simd256_u32_langext_t ored = a | b; @@ -841,13 +841,13 @@ namespace bx result.lo = simd128_or(_a.lo, _b.lo); result.hi = simd128_or(_a.hi, _b.hi); return result; -#endif // BX_SIMD_LANGEXT +#endif // BX_SIMD256_LANGEXT } template<> inline BX_CONSTEXPR_FUNC simd256_ref_t simd256_xor(simd256_ref_t _a, simd256_ref_t _b) { -#if BX_SIMD_LANGEXT +#if BX_SIMD256_LANGEXT const simd256_u32_langext_t a = bitCast(_a); const simd256_u32_langext_t b = bitCast(_b); const simd256_u32_langext_t xored = a ^ b; @@ -858,7 +858,7 @@ namespace bx result.lo = simd128_xor(_a.lo, _b.lo); result.hi = simd128_xor(_a.hi, _b.hi); return result; -#endif // BX_SIMD_LANGEXT +#endif // BX_SIMD256_LANGEXT } template<> @@ -870,7 +870,7 @@ namespace bx template<> inline BX_CONST_FUNC simd256_ref_t simd256_andc(simd256_ref_t _a, simd256_ref_t _b) { -#if BX_SIMD_LANGEXT +#if BX_SIMD256_LANGEXT const simd256_u32_langext_t a = bitCast(_a); const simd256_u32_langext_t b = bitCast(_b); const simd256_u32_langext_t notb = ~b; @@ -879,7 +879,7 @@ namespace bx return result; #else return simd_andc_ni(_a, _b); -#endif // BX_SIMD_LANGEXT +#endif // BX_SIMD256_LANGEXT } template<> @@ -903,7 +903,7 @@ namespace bx template<> inline BX_CONSTEXPR_FUNC simd256_ref_t simd256_x32_sll(simd256_ref_t _a, int _count) { -#if BX_SIMD_LANGEXT +#if BX_SIMD256_LANGEXT const simd256_u32_langext_t a = bitCast(_a); const simd256_u32_langext_t shifted = a << _count; const simd256_ref_t result = bitCast(shifted); @@ -913,13 +913,13 @@ namespace bx result.lo = simd128_x32_sll(_a.lo, _count); result.hi = simd128_x32_sll(_a.hi, _count); return result; -#endif // BX_SIMD_LANGEXT +#endif // BX_SIMD256_LANGEXT } template<> inline BX_CONSTEXPR_FUNC simd256_ref_t simd256_x32_srl(simd256_ref_t _a, int _count) { -#if BX_SIMD_LANGEXT +#if BX_SIMD256_LANGEXT const simd256_u32_langext_t a = bitCast(_a); const simd256_u32_langext_t shifted = a >> _count; const simd256_ref_t result = bitCast(shifted); @@ -929,13 +929,13 @@ namespace bx result.lo = simd128_x32_srl(_a.lo, _count); result.hi = simd128_x32_srl(_a.hi, _count); return result; -#endif // BX_SIMD_LANGEXT +#endif // BX_SIMD256_LANGEXT } template<> inline BX_CONSTEXPR_FUNC simd256_ref_t simd256_x32_sra(simd256_ref_t _a, int _count) { -#if BX_SIMD_LANGEXT +#if BX_SIMD256_LANGEXT const simd256_i32_langext_t a = bitCast(_a); const simd256_i32_langext_t shifted = a >> _count; const simd256_ref_t result = bitCast(shifted); @@ -945,7 +945,7 @@ namespace bx result.lo = simd128_x32_sra(_a.lo, _count); result.hi = simd128_x32_sra(_a.hi, _count); return result; -#endif // BX_SIMD_LANGEXT +#endif // BX_SIMD256_LANGEXT } template<> @@ -981,7 +981,7 @@ namespace bx template<> inline BX_CONSTEXPR_FUNC simd256_ref_t simd256_f64_add(simd256_ref_t _a, simd256_ref_t _b) { -#if BX_SIMD_LANGEXT +#if BX_SIMD256_LANGEXT const simd256_f64_langext_t a = bitCast(_a); const simd256_f64_langext_t b = bitCast(_b); const simd256_f64_langext_t sum = a + b; @@ -992,13 +992,13 @@ namespace bx result.lo = simd128_f64_add(_a.lo, _b.lo); result.hi = simd128_f64_add(_a.hi, _b.hi); return result; -#endif // BX_SIMD_LANGEXT +#endif // BX_SIMD256_LANGEXT } template<> inline BX_CONSTEXPR_FUNC simd256_ref_t simd256_f64_sub(simd256_ref_t _a, simd256_ref_t _b) { -#if BX_SIMD_LANGEXT +#if BX_SIMD256_LANGEXT const simd256_f64_langext_t a = bitCast(_a); const simd256_f64_langext_t b = bitCast(_b); const simd256_f64_langext_t diff = a - b; @@ -1009,13 +1009,13 @@ namespace bx result.lo = simd128_f64_sub(_a.lo, _b.lo); result.hi = simd128_f64_sub(_a.hi, _b.hi); return result; -#endif // BX_SIMD_LANGEXT +#endif // BX_SIMD256_LANGEXT } template<> inline BX_CONSTEXPR_FUNC simd256_ref_t simd256_f64_mul(simd256_ref_t _a, simd256_ref_t _b) { -#if BX_SIMD_LANGEXT +#if BX_SIMD256_LANGEXT const simd256_f64_langext_t a = bitCast(_a); const simd256_f64_langext_t b = bitCast(_b); const simd256_f64_langext_t prod = a * b; @@ -1026,13 +1026,13 @@ namespace bx result.lo = simd128_f64_mul(_a.lo, _b.lo); result.hi = simd128_f64_mul(_a.hi, _b.hi); return result; -#endif // BX_SIMD_LANGEXT +#endif // BX_SIMD256_LANGEXT } template<> inline BX_CONSTEXPR_FUNC simd256_ref_t simd256_f64_div(simd256_ref_t _a, simd256_ref_t _b) { -#if BX_SIMD_LANGEXT +#if BX_SIMD256_LANGEXT const simd256_f64_langext_t a = bitCast(_a); const simd256_f64_langext_t b = bitCast(_b); const simd256_f64_langext_t quot = a / b; @@ -1043,7 +1043,7 @@ namespace bx result.lo = simd128_f64_div(_a.lo, _b.lo); result.hi = simd128_f64_div(_a.hi, _b.hi); return result; -#endif // BX_SIMD_LANGEXT +#endif // BX_SIMD256_LANGEXT } template<> @@ -1067,7 +1067,7 @@ namespace bx template<> inline BX_CONSTEXPR_FUNC simd256_ref_t simd256_f64_madd(simd256_ref_t _a, simd256_ref_t _b, simd256_ref_t _c) { -#if BX_SIMD_LANGEXT +#if BX_SIMD256_LANGEXT const simd256_f64_langext_t a = bitCast(_a); const simd256_f64_langext_t b = bitCast(_b); const simd256_f64_langext_t c = bitCast(_c); @@ -1080,13 +1080,13 @@ namespace bx result.lo = simd128_f64_madd(_a.lo, _b.lo, _c.lo); result.hi = simd128_f64_madd(_a.hi, _b.hi, _c.hi); return result; -#endif // BX_SIMD_LANGEXT +#endif // BX_SIMD256_LANGEXT } template<> inline BX_CONSTEXPR_FUNC simd256_ref_t simd256_f64_nmsub(simd256_ref_t _a, simd256_ref_t _b, simd256_ref_t _c) { -#if BX_SIMD_LANGEXT +#if BX_SIMD256_LANGEXT const simd256_f64_langext_t a = bitCast(_a); const simd256_f64_langext_t b = bitCast(_b); const simd256_f64_langext_t c = bitCast(_c); @@ -1099,13 +1099,13 @@ namespace bx result.lo = simd128_f64_nmsub(_a.lo, _b.lo, _c.lo); result.hi = simd128_f64_nmsub(_a.hi, _b.hi, _c.hi); return result; -#endif // BX_SIMD_LANGEXT +#endif // BX_SIMD256_LANGEXT } template<> inline BX_CONSTEXPR_FUNC simd256_ref_t simd256_f64_neg(simd256_ref_t _a) { -#if BX_SIMD_LANGEXT +#if BX_SIMD256_LANGEXT const simd256_f64_langext_t a = bitCast(_a); const simd256_f64_langext_t neg = -a; const simd256_ref_t result = bitCast(neg); @@ -1115,7 +1115,7 @@ namespace bx result.lo = simd128_f64_neg(_a.lo); result.hi = simd128_f64_neg(_a.hi); return result; -#endif // BX_SIMD_LANGEXT +#endif // BX_SIMD256_LANGEXT } template<> @@ -1138,7 +1138,7 @@ namespace bx template<> inline BX_CONSTEXPR_FUNC simd256_ref_t simd256_f64_lerp(simd256_ref_t _a, simd256_ref_t _b, simd256_ref_t _s) { -#if BX_SIMD_LANGEXT +#if BX_SIMD256_LANGEXT const simd256_f64_langext_t a = bitCast(_a); const simd256_f64_langext_t b = bitCast(_b); const simd256_f64_langext_t s = bitCast(_s); @@ -1152,7 +1152,7 @@ namespace bx result.lo = simd128_f64_lerp(_a.lo, _b.lo, _s.lo); result.hi = simd128_f64_lerp(_a.hi, _b.hi, _s.hi); return result; -#endif // BX_SIMD_LANGEXT +#endif // BX_SIMD256_LANGEXT } template<> @@ -1212,7 +1212,7 @@ namespace bx template<> inline BX_CONSTEXPR_FUNC simd256_ref_t simd256_f64_cmpeq(simd256_ref_t _a, simd256_ref_t _b) { -#if BX_SIMD_LANGEXT +#if BX_SIMD256_LANGEXT const simd256_f64_langext_t a = bitCast(_a); const simd256_f64_langext_t b = bitCast(_b); const simd256_f64_langext_t cmp = a == b; @@ -1223,13 +1223,13 @@ namespace bx result.lo = simd128_f64_cmpeq(_a.lo, _b.lo); result.hi = simd128_f64_cmpeq(_a.hi, _b.hi); return result; -#endif // BX_SIMD_LANGEXT +#endif // BX_SIMD256_LANGEXT } template<> inline BX_CONSTEXPR_FUNC simd256_ref_t simd256_f64_cmpneq(simd256_ref_t _a, simd256_ref_t _b) { -#if BX_SIMD_LANGEXT +#if BX_SIMD256_LANGEXT const simd256_f64_langext_t a = bitCast(_a); const simd256_f64_langext_t b = bitCast(_b); const simd256_f64_langext_t cmp = a != b; @@ -1240,13 +1240,13 @@ namespace bx result.lo = simd128_f64_cmpneq(_a.lo, _b.lo); result.hi = simd128_f64_cmpneq(_a.hi, _b.hi); return result; -#endif // BX_SIMD_LANGEXT +#endif // BX_SIMD256_LANGEXT } template<> inline BX_CONSTEXPR_FUNC simd256_ref_t simd256_f64_cmplt(simd256_ref_t _a, simd256_ref_t _b) { -#if BX_SIMD_LANGEXT +#if BX_SIMD256_LANGEXT const simd256_f64_langext_t a = bitCast(_a); const simd256_f64_langext_t b = bitCast(_b); const simd256_f64_langext_t cmp = a < b; @@ -1257,13 +1257,13 @@ namespace bx result.lo = simd128_f64_cmplt(_a.lo, _b.lo); result.hi = simd128_f64_cmplt(_a.hi, _b.hi); return result; -#endif // BX_SIMD_LANGEXT +#endif // BX_SIMD256_LANGEXT } template<> inline BX_CONSTEXPR_FUNC simd256_ref_t simd256_f64_cmple(simd256_ref_t _a, simd256_ref_t _b) { -#if BX_SIMD_LANGEXT +#if BX_SIMD256_LANGEXT const simd256_f64_langext_t a = bitCast(_a); const simd256_f64_langext_t b = bitCast(_b); const simd256_f64_langext_t cmp = a <= b; @@ -1274,13 +1274,13 @@ namespace bx result.lo = simd128_f64_cmple(_a.lo, _b.lo); result.hi = simd128_f64_cmple(_a.hi, _b.hi); return result; -#endif // BX_SIMD_LANGEXT +#endif // BX_SIMD256_LANGEXT } template<> inline BX_CONSTEXPR_FUNC simd256_ref_t simd256_f64_cmpgt(simd256_ref_t _a, simd256_ref_t _b) { -#if BX_SIMD_LANGEXT +#if BX_SIMD256_LANGEXT const simd256_f64_langext_t a = bitCast(_a); const simd256_f64_langext_t b = bitCast(_b); const simd256_f64_langext_t cmp = a > b; @@ -1291,13 +1291,13 @@ namespace bx result.lo = simd128_f64_cmpgt(_a.lo, _b.lo); result.hi = simd128_f64_cmpgt(_a.hi, _b.hi); return result; -#endif // BX_SIMD_LANGEXT +#endif // BX_SIMD256_LANGEXT } template<> inline BX_CONSTEXPR_FUNC simd256_ref_t simd256_f64_cmpge(simd256_ref_t _a, simd256_ref_t _b) { -#if BX_SIMD_LANGEXT +#if BX_SIMD256_LANGEXT const simd256_f64_langext_t a = bitCast(_a); const simd256_f64_langext_t b = bitCast(_b); const simd256_f64_langext_t cmp = a >= b; @@ -1308,13 +1308,13 @@ namespace bx result.lo = simd128_f64_cmpge(_a.lo, _b.lo); result.hi = simd128_f64_cmpge(_a.hi, _b.hi); return result; -#endif // BX_SIMD_LANGEXT +#endif // BX_SIMD256_LANGEXT } template<> inline BX_CONSTEXPR_FUNC simd256_ref_t simd256_i64_add(simd256_ref_t _a, simd256_ref_t _b) { -#if BX_SIMD_LANGEXT +#if BX_SIMD256_LANGEXT const simd256_i64_langext_t a = bitCast(_a); const simd256_i64_langext_t b = bitCast(_b); const simd256_i64_langext_t sum = a + b; @@ -1325,13 +1325,13 @@ namespace bx result.lo = simd128_i64_add(_a.lo, _b.lo); result.hi = simd128_i64_add(_a.hi, _b.hi); return result; -#endif // BX_SIMD_LANGEXT +#endif // BX_SIMD256_LANGEXT } template<> inline BX_CONSTEXPR_FUNC simd256_ref_t simd256_i64_sub(simd256_ref_t _a, simd256_ref_t _b) { -#if BX_SIMD_LANGEXT +#if BX_SIMD256_LANGEXT const simd256_i64_langext_t a = bitCast(_a); const simd256_i64_langext_t b = bitCast(_b); const simd256_i64_langext_t diff = a - b; @@ -1342,13 +1342,13 @@ namespace bx result.lo = simd128_i64_sub(_a.lo, _b.lo); result.hi = simd128_i64_sub(_a.hi, _b.hi); return result; -#endif // BX_SIMD_LANGEXT +#endif // BX_SIMD256_LANGEXT } template<> inline BX_CONSTEXPR_FUNC simd256_ref_t simd256_u64_add(simd256_ref_t _a, simd256_ref_t _b) { -#if BX_SIMD_LANGEXT +#if BX_SIMD256_LANGEXT const simd256_u64_langext_t a = bitCast(_a); const simd256_u64_langext_t b = bitCast(_b); const simd256_u64_langext_t sum = a + b; @@ -1359,13 +1359,13 @@ namespace bx result.lo = simd128_u64_add(_a.lo, _b.lo); result.hi = simd128_u64_add(_a.hi, _b.hi); return result; -#endif // BX_SIMD_LANGEXT +#endif // BX_SIMD256_LANGEXT } template<> inline BX_CONSTEXPR_FUNC simd256_ref_t simd256_u64_sub(simd256_ref_t _a, simd256_ref_t _b) { -#if BX_SIMD_LANGEXT +#if BX_SIMD256_LANGEXT const simd256_u64_langext_t a = bitCast(_a); const simd256_u64_langext_t b = bitCast(_b); const simd256_u64_langext_t diff = a - b; @@ -1376,13 +1376,13 @@ namespace bx result.lo = simd128_u64_sub(_a.lo, _b.lo); result.hi = simd128_u64_sub(_a.hi, _b.hi); return result; -#endif // BX_SIMD_LANGEXT +#endif // BX_SIMD256_LANGEXT } template<> inline BX_CONSTEXPR_FUNC simd256_ref_t simd256_i16_add(simd256_ref_t _a, simd256_ref_t _b) { -#if BX_SIMD_LANGEXT +#if BX_SIMD256_LANGEXT const simd256_i16_langext_t a = bitCast(_a); const simd256_i16_langext_t b = bitCast(_b); const simd256_i16_langext_t sum = a + b; @@ -1393,13 +1393,13 @@ namespace bx result.lo = simd128_i16_add(_a.lo, _b.lo); result.hi = simd128_i16_add(_a.hi, _b.hi); return result; -#endif // BX_SIMD_LANGEXT +#endif // BX_SIMD256_LANGEXT } template<> inline BX_CONSTEXPR_FUNC simd256_ref_t simd256_i16_sub(simd256_ref_t _a, simd256_ref_t _b) { -#if BX_SIMD_LANGEXT +#if BX_SIMD256_LANGEXT const simd256_i16_langext_t a = bitCast(_a); const simd256_i16_langext_t b = bitCast(_b); const simd256_i16_langext_t diff = a - b; @@ -1410,13 +1410,13 @@ namespace bx result.lo = simd128_i16_sub(_a.lo, _b.lo); result.hi = simd128_i16_sub(_a.hi, _b.hi); return result; -#endif // BX_SIMD_LANGEXT +#endif // BX_SIMD256_LANGEXT } template<> inline BX_CONSTEXPR_FUNC simd256_ref_t simd256_i16_mullo(simd256_ref_t _a, simd256_ref_t _b) { -#if BX_SIMD_LANGEXT +#if BX_SIMD256_LANGEXT const simd256_i16_langext_t a = bitCast(_a); const simd256_i16_langext_t b = bitCast(_b); const simd256_i16_langext_t prod = a * b; @@ -1427,13 +1427,13 @@ namespace bx result.lo = simd128_i16_mullo(_a.lo, _b.lo); result.hi = simd128_i16_mullo(_a.hi, _b.hi); return result; -#endif // BX_SIMD_LANGEXT +#endif // BX_SIMD256_LANGEXT } template<> inline BX_CONSTEXPR_FUNC simd256_ref_t simd256_i16_cmpeq(simd256_ref_t _a, simd256_ref_t _b) { -#if BX_SIMD_LANGEXT +#if BX_SIMD256_LANGEXT const simd256_i16_langext_t a = bitCast(_a); const simd256_i16_langext_t b = bitCast(_b); const simd256_i16_langext_t cmp = a == b; @@ -1444,13 +1444,13 @@ namespace bx result.lo = simd128_i16_cmpeq(_a.lo, _b.lo); result.hi = simd128_i16_cmpeq(_a.hi, _b.hi); return result; -#endif // BX_SIMD_LANGEXT +#endif // BX_SIMD256_LANGEXT } template<> inline BX_CONSTEXPR_FUNC simd256_ref_t simd256_x16_sll(simd256_ref_t _a, int _count) { -#if BX_SIMD_LANGEXT +#if BX_SIMD256_LANGEXT const simd256_u16_langext_t a = bitCast(_a); const simd256_u16_langext_t shifted = a << _count; const simd256_ref_t result = bitCast(shifted); @@ -1460,13 +1460,13 @@ namespace bx result.lo = simd128_x16_sll(_a.lo, _count); result.hi = simd128_x16_sll(_a.hi, _count); return result; -#endif // BX_SIMD_LANGEXT +#endif // BX_SIMD256_LANGEXT } template<> inline BX_CONSTEXPR_FUNC simd256_ref_t simd256_x16_srl(simd256_ref_t _a, int _count) { -#if BX_SIMD_LANGEXT +#if BX_SIMD256_LANGEXT const simd256_u16_langext_t a = bitCast(_a); const simd256_u16_langext_t shifted = a >> _count; const simd256_ref_t result = bitCast(shifted); @@ -1476,13 +1476,13 @@ namespace bx result.lo = simd128_x16_srl(_a.lo, _count); result.hi = simd128_x16_srl(_a.hi, _count); return result; -#endif // BX_SIMD_LANGEXT +#endif // BX_SIMD256_LANGEXT } template<> inline BX_CONSTEXPR_FUNC simd256_ref_t simd256_i8_add(simd256_ref_t _a, simd256_ref_t _b) { -#if BX_SIMD_LANGEXT +#if BX_SIMD256_LANGEXT const simd256_i8_langext_t a = bitCast(_a); const simd256_i8_langext_t b = bitCast(_b); const simd256_i8_langext_t sum = a + b; @@ -1493,13 +1493,13 @@ namespace bx result.lo = simd128_i8_add(_a.lo, _b.lo); result.hi = simd128_i8_add(_a.hi, _b.hi); return result; -#endif // BX_SIMD_LANGEXT +#endif // BX_SIMD256_LANGEXT } template<> inline BX_CONSTEXPR_FUNC simd256_ref_t simd256_i8_sub(simd256_ref_t _a, simd256_ref_t _b) { -#if BX_SIMD_LANGEXT +#if BX_SIMD256_LANGEXT const simd256_i8_langext_t a = bitCast(_a); const simd256_i8_langext_t b = bitCast(_b); const simd256_i8_langext_t diff = a - b; @@ -1510,7 +1510,7 @@ namespace bx result.lo = simd128_i8_sub(_a.lo, _b.lo); result.hi = simd128_i8_sub(_a.hi, _b.hi); return result; -#endif // BX_SIMD_LANGEXT +#endif // BX_SIMD256_LANGEXT } template<> diff --git a/include/bx/simd_t.h b/include/bx/simd_t.h index d92f7c3..75b9299 100644 --- a/include/bx/simd_t.h +++ b/include/bx/simd_t.h @@ -113,6 +113,12 @@ extern "C" float fmaf(float, float, float); # define BX_SIMD_SUPPORTED 1 #endif // BX_SIMD_* +#if BX_SIMD_LANGEXT && BX_SIMD_AVX +# define BX_SIMD256_LANGEXT 1 +#else +# define BX_SIMD256_LANGEXT 0 +#endif // BX_SIMD_LANGEXT && BX_SIMD_AVX + #define BX_SIMD_FORCE_INLINE BX_FORCE_INLINE #define BX_SIMD_INLINE inline diff --git a/tests/math_test.cpp b/tests/math_test.cpp index eaddf62..d132020 100644 --- a/tests/math_test.cpp +++ b/tests/math_test.cpp @@ -712,7 +712,7 @@ TEST_CASE("exp", "[math][libm]") { STATIC_REQUIRE( 1.0f == bx::exp(-0.0f) ); STATIC_REQUIRE( 0.0f == bx::exp(-bx::kFloatInfinity) ); - STATIC_REQUIRE( 0.0f == bx::exp(bx::log(bx::kFloatSmallest) ) ); + STATIC_REQUIRE(bx::exp(bx::log(bx::kFloatSmallest) ) <= bx::kFloatSmallest); bx::WriterI* writer = bx::getNullOut(); bx::Error err; diff --git a/tests/scanner_test.cpp b/tests/scanner_test.cpp index ddafdc7..db79149 100644 --- a/tests/scanner_test.cpp +++ b/tests/scanner_test.cpp @@ -25,6 +25,7 @@ namespace bx { StringView sv = scanner.acceptUntil(Scanner::Class::EndOfLine); BX_TRACE("%d: '%S'", scanner.getLine(), &sv); + BX_UNUSED(sv); scanner.acceptUntil(Scanner::Class::NewLine); }