new files for v2.5

This commit is contained in:
Richard Geldreich
2026-07-01 13:20:12 -04:00
parent 56ba46d485
commit 3cfdd2240b
13 changed files with 16769 additions and 0 deletions

File diff suppressed because it is too large Load Diff

View File

@@ -0,0 +1,440 @@
// basisu_astc_ldr_fencode.h
#pragma once
#include "../transcoder/basisu.h"
#include "../transcoder/basisu_transcoder_internal.h"
#include "basisu_astc_ldr_common.h"
namespace basisu
{
namespace astc_ldrf
{
BASISU_FORCE_INLINE int popcount64(uint64_t x)
{
#if defined(__cplusplus) && (__cplusplus >= 202002L) && defined(__cpp_lib_bitops)
return static_cast<int>(std::popcount(x));
#elif defined(__EMSCRIPTEN__) || defined(__clang__) || defined(__GNUC__)
return __builtin_popcountll(static_cast<unsigned long long>(x));
#elif defined(_MSC_VER) && defined(_M_X64)
return static_cast<int>(__popcnt64(x));
#elif defined(_MSC_VER) && (defined(_M_IX86) || defined(_M_ARM64) || defined(_M_ARM))
return static_cast<int>(
__popcnt(static_cast<uint32_t>(x)) +
__popcnt(static_cast<uint32_t>(x >> 32))
);
#else
int count = 0;
while (x)
{
x &= (x - 1);
++count;
}
return count;
#endif
}
struct bitmask192
{
uint64_t m_a, m_b, m_c; // a=lowest qword (bits 0-63), b=middle (64-127), c=highest (128-191)
inline bitmask192() {}
constexpr inline bitmask192(const bitmask192& o) noexcept
: m_a(o.m_a), m_b(o.m_b), m_c(o.m_c)
{
}
inline bitmask192& operator=(const bitmask192& o) noexcept
{
m_a = o.m_a;
m_b = o.m_b;
m_c = o.m_c;
return *this;
}
constexpr inline bitmask192(uint64_t a, uint64_t b, uint64_t c) noexcept
: m_a(a), m_b(b), m_c(c)
{
}
static constexpr inline bitmask192 zero() noexcept
{
return bitmask192(0, 0, 0);
}
static constexpr inline bitmask192 all() noexcept
{
return bitmask192(UINT64_MAX, UINT64_MAX, UINT64_MAX);
}
inline void clear() noexcept
{
m_a = 0;
m_b = 0;
m_c = 0;
}
inline void set_all() noexcept
{
m_a = UINT64_MAX;
m_b = UINT64_MAX;
m_c = UINT64_MAX;
}
inline bool is_zero() const noexcept
{
return (m_a | m_b | m_c) == 0;
}
inline bool any() const noexcept
{
return (m_a | m_b | m_c) != 0;
}
inline bool none() const noexcept
{
return !any();
}
inline bool all_set() const noexcept
{
return (m_a == UINT64_MAX) && (m_b == UINT64_MAX) && (m_c == UINT64_MAX);
}
inline uint32_t popcount() const noexcept
{
return popcount64(m_a) + popcount64(m_b) + popcount64(m_c);
}
constexpr inline bitmask192 operator~() const noexcept
{
return bitmask192(~m_a, ~m_b, ~m_c);
}
constexpr inline bitmask192 operator&(const bitmask192& rhs) const noexcept
{
return bitmask192(m_a & rhs.m_a, m_b & rhs.m_b, m_c & rhs.m_c);
}
constexpr inline bitmask192 operator|(const bitmask192& rhs) const noexcept
{
return bitmask192(m_a | rhs.m_a, m_b | rhs.m_b, m_c | rhs.m_c);
}
constexpr inline bitmask192 operator^(const bitmask192& rhs) const noexcept
{
return bitmask192(m_a ^ rhs.m_a, m_b ^ rhs.m_b, m_c ^ rhs.m_c);
}
inline bitmask192& operator&=(const bitmask192& rhs) noexcept
{
m_a &= rhs.m_a;
m_b &= rhs.m_b;
m_c &= rhs.m_c;
return *this;
}
inline bitmask192& operator|=(const bitmask192& rhs) noexcept
{
m_a |= rhs.m_a;
m_b |= rhs.m_b;
m_c |= rhs.m_c;
return *this;
}
inline bitmask192& operator^=(const bitmask192& rhs) noexcept
{
m_a ^= rhs.m_a;
m_b ^= rhs.m_b;
m_c ^= rhs.m_c;
return *this;
}
constexpr inline bool operator==(const bitmask192& rhs) const noexcept
{
return (m_a == rhs.m_a) && (m_b == rhs.m_b) && (m_c == rhs.m_c);
}
constexpr inline bool operator!=(const bitmask192& rhs) const noexcept
{
return !(*this == rhs);
}
// m_a is the lowest qword, m_b highest
constexpr inline bool operator<(const bitmask192& rhs) const noexcept
{
if (m_c != rhs.m_c) return m_c < rhs.m_c;
if (m_b != rhs.m_b) return m_b < rhs.m_b;
return m_a < rhs.m_a;
}
constexpr inline bool operator>(const bitmask192& rhs) const noexcept
{
return rhs < *this;
}
constexpr inline bool operator<=(const bitmask192& rhs) const noexcept
{
return !(rhs < *this);
}
constexpr inline bool operator>=(const bitmask192& rhs) const noexcept
{
return !(*this < rhs);
}
explicit constexpr inline operator bool() const noexcept
{
return (m_a | m_b | m_c) != 0;
}
inline bool intersects(const bitmask192& rhs) const noexcept
{
return ((m_a & rhs.m_a) | (m_b & rhs.m_b) | (m_c & rhs.m_c)) != 0;
}
inline bool disjoint(const bitmask192& rhs) const noexcept
{
return !intersects(rhs);
}
constexpr inline bool contains_all(const bitmask192& rhs) const noexcept
{
return ((m_a & rhs.m_a) == rhs.m_a) &&
((m_b & rhs.m_b) == rhs.m_b) &&
((m_c & rhs.m_c) == rhs.m_c);
}
constexpr inline bool is_subset_of(const bitmask192& rhs) const noexcept
{
return rhs.contains_all(*this);
}
inline bool contains_any(const bitmask192& rhs) const noexcept
{
return intersects(rhs);
}
constexpr inline uint64_t low64() const noexcept
{
return m_a;
}
constexpr inline uint64_t mid64() const noexcept
{
return m_b;
}
constexpr inline uint64_t high64() const noexcept
{
return m_c;
}
inline void set_bit(uint32_t index) noexcept
{
assert(index < 192);
if (index < 64)
m_a |= uint64_t(1) << index;
else if (index < 128)
m_b |= uint64_t(1) << (index - 64);
else
m_c |= uint64_t(1) << (index - 128);
}
inline void set_bit(uint32_t index, uint32_t bit_val) noexcept
{
assert(index < 192);
assert(bit_val <= 1);
if (index < 64)
{
m_a &= ~(uint64_t(1) << index);
m_a |= uint64_t(bit_val) << index;
}
else if (index < 128)
{
m_b &= ~(uint64_t(1) << (index - 64));
m_b |= uint64_t(bit_val) << (index - 64);
}
else
{
m_c &= ~(uint64_t(1) << (index - 128));
m_c |= uint64_t(bit_val) << (index - 128);
}
}
inline bool is_bit_set(uint32_t index) const noexcept
{
assert(index < 192);
if (index < 64)
return ((m_a >> index) & 1) != 0;
else if (index < 128)
return ((m_b >> (index - 64)) & 1) != 0;
else
return ((m_c >> (index - 128)) & 1) != 0;
}
static inline bitmask192 lsb_mask(uint32_t num_bits) noexcept
{
assert(num_bits <= 192);
if (num_bits == 0)
return bitmask192(0, 0, 0);
if (num_bits < 64)
return bitmask192((uint64_t(1) << num_bits) - 1, 0, 0);
if (num_bits == 64)
return bitmask192(UINT64_MAX, 0, 0);
if (num_bits < 128)
{
return bitmask192(
UINT64_MAX,
(uint64_t(1) << (num_bits - 64)) - 1,
0);
}
if (num_bits == 128)
return bitmask192(UINT64_MAX, UINT64_MAX, 0);
if (num_bits < 192)
{
return bitmask192(
UINT64_MAX,
UINT64_MAX,
(uint64_t(1) << (num_bits - 128)) - 1);
}
return bitmask192(UINT64_MAX, UINT64_MAX, UINT64_MAX);
}
};
inline uint32_t popcount192(const bitmask192& a) { return a.popcount(); }
const uint32_t MAX_CANDIDATES = 512;
struct rgba32_image
{
const uint8_t* m_pPixels;
uint32_t m_width;
uint32_t m_height;
uint32_t m_row_pitch_in_texels; // pitch in pixels/texels, not bytes
};
struct single_subset_enc_context
{
uint32_t m_block_width, m_block_height;
uint32_t m_block_size_index;
uint32_t m_total_block_pixels;
uint32_t m_max_candidates;
uint32_t m_num_ls_iterations;
uint32_t m_chan_weights[4];
basist::astc_ldr_t::dct2f m_dct;
astc_helpers::decode_mode m_astc_decode_mode;
bool m_disable_dual_plane;
bool m_weight_polishing;
bool m_has_alpha;
bool m_try_base_ofs;
bool m_higher_effort_bc;
};
const uint32_t MAX_UNIQUE_2SUBSET_PATS = 838;
const uint32_t MAX_UNIQUE_3SUBSET_PATS = 626;
typedef basisu::vector<astc_helpers::log_astc_block> astc_lblock_vec;
// source pEndpoints[] = ASTC direct order: LR HR LG HG LB HB LA HA
bool cem_encode(uint32_t cem_index, const float pEndpoints[8], uint32_t endpoint_ise_range, uint8_t* pCEM_values, bool allow_bc = true, bool high_effort = true);
bool init_single_subset_context(
single_subset_enc_context& ctx,
uint32_t block_width, uint32_t block_height,
astc_helpers::decode_mode astc_decode_mode,
const uint32_t chan_weights[4],
uint32_t max_candidates, uint32_t num_ls_iterations, bool disable_dual_plane, bool has_alpha, bool weight_polishing);
// best_lblock will be in ise space (see is_lblock_ise() below), elements in array pointed to by pAll_candidates (which may be null) will be in rank space
double compress_single_subset(
single_subset_enc_context& ctx,
const uint8_t* pBlock_pixels,
astc_helpers::log_astc_block& best_lblock,
astc_lblock_vec* pAll_candidates,
bool always_compute_error);
static const uint32_t NUM_DOT_THRESH_FRACTS = 6;
struct subset_enc_context : single_subset_enc_context
{
subset_enc_context() {}
uint32_t m_max_subsets;
uint32_t m_num_carrier_candidates;
uint32_t m_num_pattern_candidates;
float m_two_subset_var_thresh;
float m_three_subset_var_thresh;
uint32_t m_two_subset_dot_thresh_fract_index;
bitmask192 m_two_subset_pat_bitmask[MAX_UNIQUE_2SUBSET_PATS];
uint32_t m_num_unique_two_subset_pats;
const uint16_t* m_pUnique_two_subset_pats;
bool m_use_method1;
bool m_use_method2;
astc_ldr::partitions_data* m_pPart_data_p2;
astc_ldr::partitions_data* m_pPart_data_p3;
};
// must have first called init_single_subset_context() on ctx
bool init_multi_subset_context(
subset_enc_context& ctx,
uint32_t max_subsets,
uint32_t num_carrier_candidates, uint32_t num_pattern_candidates,
float two_subset_var_thresh, uint32_t two_subset_dot_thresh_fract_index,
float three_subset_var_thresh,
astc_ldr::partitions_data* pPart_data_p2, astc_ldr::partitions_data* pPart_data_p3);
struct subset_enc_thread_context
{
astc_ldr::partition_pattern_vec m_pat_vec;
};
double compress_block_subsets(
const subset_enc_context& enc_context,
subset_enc_thread_context& enc_thread_context,
const uint8_t* pBlock_pixels,
astc_helpers::log_astc_block& best_lblock, astc_lblock_vec* pAll_candidates);
enum
{
cUserModeISEValues = 0,
cUserModeRankValues = 1
};
static inline bool is_lblock_ise(const astc_helpers::log_astc_block& log_blk)
{
// the default user mode value == ISE, 1 = ranks (which is the exceptional case for astc_helpers)
return log_blk.m_user_mode == cUserModeISEValues;
}
void convert_rank_lblock_to_ise(astc_helpers::log_astc_block& log_blk);
void convert_ise_lblock_to_rank(astc_helpers::log_astc_block& log_blk);
} // namespace astc_ldr_f
} // namespace basisu

View File

@@ -0,0 +1,298 @@
0.6883562f, -0.1883562f, 0.4143836f, 0.0856164f, 0.0856164f, 0.4143836f, -0.1883562f, 0.6883562f,
0.9555160f, -0.2272727f, 0.0444840f, 0.1423488f, 0.7272727f, -0.1423488f, -0.1423488f, 0.7272727f,
0.1423488f, 0.0444840f, -0.2272727f, 0.9555160f, 0.6000000f, -0.2000000f, 0.4000000f, 0.0000000f,
0.2000000f, 0.2000000f, 0.0000000f, 0.4000000f, -0.2000000f, 0.6000000f, 0.8285714f, -0.1428571f,
0.0285714f, 0.3428572f, 0.2857143f, -0.0571429f, -0.1428571f, 0.7142857f, -0.1428571f, -0.0571429f,
0.2857143f, 0.3428572f, 0.0285714f, -0.1428571f, 0.8285714f, 0.9857143f, -0.2523810f, 0.0809524f,
-0.0142857f, 0.0571429f, 1.0095239f, -0.3238095f, 0.0571429f, -0.0857143f, 0.4857143f, 0.4857143f,
-0.0857143f, 0.0571429f, -0.3238095f, 1.0095239f, 0.0571429f, -0.0142857f, 0.0809524f, -0.2523810f,
0.9857143f, 0.5107527f, -0.1774193f, 0.3817204f, -0.0483871f, 0.2526882f, 0.0806452f, 0.0806452f,
0.2526882f, -0.0483871f, 0.3817204f, -0.1774193f, 0.5107527f, 0.7542282f, -0.1948819f, 0.0528584f,
0.3983119f, 0.1476378f, -0.0400442f, -0.0169237f, 0.5472441f, -0.1484306f, -0.1484306f, 0.5472441f,
-0.0169237f, -0.0400442f, 0.1476378f, 0.3983119f, 0.0528584f, -0.1948819f, 0.7542282f, 0.9212349f,
-0.2166766f, 0.0636154f, -0.0130717f, 0.2100402f, 0.5778043f, -0.1696410f, 0.0348577f, -0.1641219f,
0.7987264f, -0.0538284f, 0.0110606f, 0.0110606f, -0.0538284f, 0.7987264f, -0.1641219f, 0.0348577f,
-0.1696410f, 0.5778043f, 0.2100402f, -0.0130717f, 0.0636154f, -0.2166766f, 0.9212349f, 0.9969320f,
-0.2099229f, 0.0692308f, -0.0208463f, 0.0030679f, 0.0163624f, 1.1195889f, -0.3692308f, 0.1111804f,
-0.0163624f, -0.0354519f, 0.2408908f, 0.8000000f, -0.2408908f, 0.0354519f, 0.0354519f, -0.2408908f,
0.8000000f, 0.2408908f, -0.0354519f, -0.0163624f, 0.1111804f, -0.3692308f, 1.1195889f, 0.0163624f,
0.0030679f, -0.0208463f, 0.0692308f, -0.2099229f, 0.9969320f, 0.4159091f, -0.1659091f, 0.3431818f,
-0.0931818f, 0.2340909f, 0.0159091f, 0.1613636f, 0.0886364f, 0.0886364f, 0.1613636f, 0.0159091f,
0.2340909f, -0.0931818f, 0.3431818f, -0.1659091f, 0.4159091f, 0.6538065f, -0.1721698f, 0.0584577f,
0.3956889f, 0.0400943f, -0.0136134f, 0.1891948f, 0.2099057f, -0.0712703f, -0.0689228f, 0.4221698f,
-0.1433414f, -0.1433414f, 0.4221698f, -0.0689228f, -0.0712703f, 0.2099057f, 0.1891948f, -0.0136134f,
0.0400943f, 0.3956889f, 0.0584577f, -0.1721698f, 0.6538065f, 0.8053632f, -0.2047127f, 0.0614056f,
-0.0163868f, 0.3634550f, 0.2204598f, -0.0661291f, 0.0176474f, -0.0784532f, 0.6456324f, -0.1936639f,
0.0516816f, -0.1215507f, 0.4554813f, 0.0815266f, -0.0217564f, -0.0217564f, 0.0815266f, 0.4554813f,
-0.1215507f, 0.0516816f, -0.1936639f, 0.6456324f, -0.0784532f, 0.0176474f, -0.0661291f, 0.2204598f,
0.3634550f, -0.0163868f, 0.0614056f, -0.2047127f, 0.8053632f, 0.8815933f, -0.2045389f, 0.0750646f,
-0.0215593f, 0.0044532f, 0.2706439f, 0.4675174f, -0.1715762f, 0.0492785f, -0.0101788f, -0.1695884f,
0.8210233f, -0.1598191f, 0.0459017f, -0.0094813f, -0.0123115f, 0.0596032f, 0.7563307f, -0.2172260f,
0.0448696f, 0.0448696f, -0.2172260f, 0.7563307f, 0.0596032f, -0.0123115f, -0.0094813f, 0.0459017f,
-0.1598191f, 0.8210233f, -0.1695884f, -0.0101788f, 0.0492785f, -0.1715762f, 0.4675174f, 0.2706439f,
0.0044532f, -0.0215593f, 0.0750646f, -0.2045389f, 0.8815933f, 0.9672751f, -0.2873513f, 0.0769021f,
-0.0186696f, 0.0054320f, -0.0009586f, 0.1047195f, 0.9195243f, -0.2460869f, 0.0597428f, -0.0173824f,
0.0030675f, -0.1279904f, 0.6539148f, 0.3007728f, -0.0730190f, 0.0212451f, -0.0037491f, 0.0649557f,
-0.3318644f, 1.0073659f, -0.1058330f, 0.0307925f, -0.0054340f, -0.0067231f, 0.0343492f, -0.1042661f,
0.9963973f, -0.2899053f, 0.0511598f, -0.0051125f, 0.0261200f, -0.0792867f, 0.3231576f, 0.5710128f,
-0.1007670f, 0.0038343f, -0.0195900f, 0.0594650f, -0.2423682f, 0.9050737f, 0.0755752f, -0.0009586f,
0.0048975f, -0.0148663f, 0.0605921f, -0.2262684f, 0.9811062f, 0.9997144f, -0.1402052f, 0.0544243f,
-0.0244578f, 0.0084673f, -0.0019448f, 0.0002094f, 0.0022848f, 1.1216413f, -0.4353943f, 0.1956628f,
-0.0677386f, 0.0155583f, -0.0016755f, -0.0063974f, 0.0594046f, 1.2191039f, -0.5478558f, 0.1896682f,
-0.0435634f, 0.0046914f, 0.0100531f, -0.0933501f, 0.3699796f, 0.8609163f, -0.2980500f, 0.0684567f,
-0.0073723f, -0.0100531f, 0.0933501f, -0.3699796f, 0.9168615f, 0.2980500f, -0.0684567f, 0.0073723f,
0.0058643f, -0.0544542f, 0.2158214f, -0.5348359f, 1.1594708f, 0.0399331f, -0.0043005f, -0.0016755f,
0.0155583f, -0.0616633f, 0.1528102f, -0.3312774f, 1.1314477f, 0.0012287f, 0.0002094f, -0.0019448f,
0.0077079f, -0.0191013f, 0.0414097f, -0.1414310f, 0.9998464f, 0.3539683f, -0.1539683f, 0.2904762f,
-0.0904762f, 0.2269841f, -0.0269841f, 0.1952381f, 0.0047619f, 0.1317460f, 0.0682540f, 0.0682540f,
0.1317460f, 0.0047619f, 0.1952381f, -0.0269841f, 0.2269841f, -0.0904762f, 0.2904762f, -0.1539683f,
0.3539683f, 0.5617812f, -0.1572600f, 0.0543796f, 0.3820210f, -0.0145911f, 0.0050455f, 0.2472007f,
0.0924105f, -0.0319550f, 0.0674404f, 0.2350793f, -0.0812891f, -0.0673798f, 0.3420810f, -0.1182897f,
-0.1308050f, 0.3438689f, -0.0590948f, -0.0911226f, 0.2395491f, 0.0666980f, -0.0382127f, 0.1004561f,
0.2344217f, 0.0146972f, -0.0386369f, 0.4021455f, 0.0543796f, -0.1429567f, 0.5279383f, 0.6755751f,
-0.1416472f, 0.0300853f, -0.0082323f, 0.4201931f, 0.0643851f, -0.0136751f, 0.0037420f, 0.1137348f,
0.3116238f, -0.0661876f, 0.0181110f, -0.1416472f, 0.5176560f, -0.1099481f, 0.0300853f, -0.0879808f,
0.3215297f, 0.0861782f, -0.0235811f, -0.0235811f, 0.0861782f, 0.3215297f, -0.0879808f, 0.0300853f,
-0.1099481f, 0.5176560f, -0.1416472f, 0.0181110f, -0.0661876f, 0.3116238f, 0.1137348f, 0.0037420f,
-0.0136751f, 0.0643851f, 0.4201931f, -0.0082323f, 0.0300853f, -0.1416472f, 0.6755751f, 0.8069659f,
-0.2107186f, 0.0594210f, -0.0171096f, 0.0045659f, 0.3617290f, 0.2269277f, -0.0639919f, 0.0184257f,
-0.0049171f, -0.0835080f, 0.6645741f, -0.1874048f, 0.0539610f, -0.0144002f, -0.1263000f, 0.4732779f,
0.1065327f, -0.0306749f, 0.0081860f, 0.0087699f, -0.0328629f, 0.6332502f, -0.1823370f, 0.0486588f,
0.0402884f, -0.1509708f, 0.5632743f, 0.0167249f, -0.0044632f, 0.0068062f, -0.0255047f, 0.0951583f,
0.4646113f, -0.1239872f, -0.0144002f, 0.0539610f, -0.2013295f, 0.6602938f, -0.0823658f, -0.0049171f,
0.0184257f, -0.0687466f, 0.2254662f, 0.3621190f, 0.0045659f, -0.0171096f, 0.0638362f, -0.2093614f,
0.8066038f, 0.8810968f, -0.2021351f, 0.0666954f, -0.0202770f, 0.0058238f, -0.0012029f, 0.2717788f,
0.4620232f, -0.1524466f, 0.0463474f, -0.0133115f, 0.0027496f, -0.1685313f, 0.8159056f, -0.1420003f,
0.0431715f, -0.0123993f, 0.0025612f, -0.0173142f, 0.0838225f, 0.6720048f, -0.2043056f, 0.0586787f,
-0.0121205f, 0.0449523f, -0.2176261f, 0.7577241f, -0.0067165f, 0.0019290f, -0.0003985f, -0.0039697f,
0.0192183f, -0.0669137f, 0.7472055f, -0.2146051f, 0.0443283f, -0.0121205f, 0.0586787f, -0.2043056f,
0.5968578f, 0.1054055f, -0.0217723f, 0.0025612f, -0.0123993f, 0.0431715f, -0.1261211f, 0.8113450f,
-0.1675893f, 0.0027496f, -0.0133115f, 0.0463474f, -0.1353993f, 0.4571270f, 0.2727902f, -0.0012029f,
0.0058238f, -0.0202770f, 0.0592372f, -0.1999931f, 0.8806543f, 0.9554276f, -0.2268213f, 0.0425861f,
-0.0101123f, 0.0019080f, -0.0005029f, 0.0001033f, 0.1426315f, 0.7258282f, -0.1362755f, 0.0323593f,
-0.0061055f, 0.0016092f, -0.0003307f, -0.1426315f, 0.7287173f, 0.1362755f, -0.0323593f, 0.0061055f,
-0.0016092f, 0.0003307f, 0.0425861f, -0.2175763f, 0.9147495f, -0.2172119f, 0.0409829f, -0.0108020f,
0.0022196f, 0.0063560f, -0.0324732f, 0.1365260f, 0.7274374f, -0.1372507f, 0.0361755f, -0.0074333f,
-0.0063560f, 0.0324732f, -0.1365260f, 0.7271081f, 0.1372507f, -0.0361755f, 0.0074333f, 0.0019080f,
-0.0097479f, 0.0409829f, -0.2182660f, 0.9193873f, -0.2423252f, 0.0497929f, 0.0002505f, -0.0012798f,
0.0053808f, -0.0286570f, 0.1207099f, 0.8116162f, -0.1667705f, -0.0002756f, 0.0014078f, -0.0059189f,
0.0315227f, -0.1327809f, 0.7072222f, 0.1834475f, 0.0001033f, -0.0005279f, 0.0022196f, -0.0118210f,
0.0497929f, -0.2652083f, 0.9312072f, 0.9899751f, -0.2765257f, 0.0928675f, -0.0243293f, 0.0065427f,
-0.0019937f, 0.0006807f, -0.0001201f, 0.0400995f, 1.1061027f, -0.3714699f, 0.0973173f, -0.0261706f,
0.0079748f, -0.0027227f, 0.0004805f, -0.0687420f, 0.3895382f, 0.6368056f, -0.1668297f, 0.0448639f,
-0.0136711f, 0.0046674f, -0.0008237f, 0.0562435f, -0.3187130f, 0.9335227f, 0.1364970f, -0.0367068f,
0.0111854f, -0.0038188f, 0.0006739f, -0.0204703f, 0.1159986f, -0.3397642f, 1.1149904f, -0.1461748f,
0.0445429f, -0.0152074f, 0.0026837f, 0.0026837f, -0.0152074f, 0.0445429f, -0.1461748f, 1.1149904f,
-0.3397642f, 0.1159986f, -0.0204703f, 0.0006739f, -0.0038188f, 0.0111854f, -0.0367068f, 0.1364970f,
0.9335227f, -0.3187130f, 0.0562435f, -0.0008237f, 0.0046674f, -0.0136711f, 0.0448639f, -0.1668297f,
0.6368056f, 0.3895382f, -0.0687420f, 0.0004805f, -0.0027227f, 0.0079748f, -0.0261706f, 0.0973173f,
-0.3714699f, 1.1061027f, 0.0400995f, -0.0001201f, 0.0006807f, -0.0019937f, 0.0065427f, -0.0243293f,
0.0928675f, -0.2765257f, 0.9899751f, 0.9999861f, -0.1427280f, 0.0322982f, -0.0127173f, 0.0061198f,
-0.0030193f, 0.0010721f, -0.0002462f, 0.0000265f, 0.0001113f, 1.1418240f, -0.2583854f, 0.1017386f,
-0.0489583f, 0.0241542f, -0.0085767f, 0.0019699f, -0.0002121f, -0.0005192f, 0.0048215f, 1.2057984f,
-0.4747804f, 0.2284722f, -0.1127197f, 0.0400247f, -0.0091929f, 0.0009900f, 0.0013500f, -0.0125358f,
0.0649243f, 1.2344289f, -0.5940278f, 0.2930712f, -0.1040642f, 0.0239017f, -0.0025740f, -0.0021214f,
0.0196992f, -0.1020238f, 0.3458974f, 0.9334723f, -0.4605404f, 0.1635294f, -0.0375598f, 0.0040449f,
0.0021214f, -0.0196992f, 0.1020238f, -0.3458974f, 0.8443055f, 0.4605404f, -0.1635294f, 0.0375598f,
-0.0040449f, -0.0014850f, 0.0137894f, -0.0714167f, 0.2421282f, -0.5910138f, 1.2776217f, 0.1144706f,
-0.0262918f, 0.0028314f, 0.0007425f, -0.0068947f, 0.0357083f, -0.1210641f, 0.2955069f, -0.6388109f,
1.2760980f, 0.0131459f, -0.0014157f, -0.0002121f, 0.0019699f, -0.0102024f, 0.0345897f, -0.0844306f,
0.1825174f, -0.3645994f, 1.1391011f, 0.0004045f, 0.0000265f, -0.0002462f, 0.0012753f, -0.0043237f,
0.0105538f, -0.0228147f, 0.0455749f, -0.1423876f, 0.9999495f, 0.2845912f, -0.1179245f, 0.2594340f,
-0.0927673f, 0.2091195f, -0.0424528f, 0.1839623f, -0.0172956f, 0.1336478f, 0.0330189f, 0.1084906f,
0.0581761f, 0.0581761f, 0.1084906f, 0.0330189f, 0.1336478f, -0.0172956f, 0.1839623f, -0.0424528f,
0.2091195f, -0.0927673f, 0.2594340f, -0.1179245f, 0.2845912f, 0.4784868f, -0.1190476f, 0.0453227f,
0.3664491f, -0.0380952f, 0.0145033f, 0.2544114f, 0.0428571f, -0.0163162f, 0.1423737f, 0.1238095f,
-0.0471356f, 0.0303360f, 0.2047619f, -0.0779550f, -0.0817017f, 0.2857143f, -0.1087745f, -0.1087745f,
0.2857143f, -0.0817017f, -0.0779550f, 0.2047619f, 0.0303360f, -0.0471356f, 0.1238095f, 0.1423737f,
-0.0163162f, 0.0428571f, 0.2544114f, 0.0145033f, -0.0380952f, 0.3664491f, 0.0453227f, -0.1190476f,
0.4784868f, 0.6098023f, -0.1552631f, 0.0402324f, -0.0131765f, 0.4185360f, 0.0020702f, -0.0005364f,
0.0001757f, 0.1794530f, 0.1987367f, -0.0514975f, 0.0168660f, -0.0118133f, 0.3560700f, -0.0922663f,
0.0302182f, -0.1430446f, 0.4367628f, -0.0855359f, 0.0280139f, -0.0819523f, 0.2502276f, 0.1009994f,
-0.0330784f, -0.0330784f, 0.1009994f, 0.2502276f, -0.0819523f, 0.0280139f, -0.0855359f, 0.4367628f,
-0.1430446f, 0.0302182f, -0.0922663f, 0.3560700f, -0.0118133f, 0.0168660f, -0.0514975f, 0.1987367f,
0.1794530f, 0.0001757f, -0.0005364f, 0.0020702f, 0.4185360f, -0.0131765f, 0.0402324f, -0.1552631f,
0.6098023f, 0.7383369f, -0.1727899f, 0.0494924f, -0.0124893f, 0.0036259f, 0.3966643f, 0.1151932f,
-0.0329949f, 0.0083262f, -0.0024173f, 0.0549918f, 0.4031764f, -0.1154822f, 0.0291417f, -0.0084605f,
-0.1588972f, 0.5473127f, -0.1175973f, 0.0296755f, -0.0086155f, -0.0755414f, 0.2601981f, 0.1996616f,
-0.0503842f, 0.0146277f, 0.0078145f, -0.0269165f, 0.5169204f, -0.1304439f, 0.0378708f, 0.0378708f,
-0.1304439f, 0.5169204f, -0.0269165f, 0.0078145f, 0.0146277f, -0.0503842f, 0.1996616f, 0.2601981f,
-0.0755414f, -0.0086155f, 0.0296755f, -0.1175973f, 0.5473127f, -0.1588972f, -0.0084605f, 0.0291417f,
-0.1154822f, 0.4031764f, 0.0549918f, -0.0024173f, 0.0083262f, -0.0329949f, 0.1151932f, 0.3966643f,
0.0036259f, -0.0124893f, 0.0494924f, -0.1727899f, 0.7383369f, 0.7979735f, -0.1758341f, 0.0515405f,
-0.0146323f, 0.0039688f, -0.0009159f, 0.3719327f, 0.2344454f, -0.0687206f, 0.0195097f, -0.0052917f,
0.0012212f, -0.1149711f, 0.7033362f, -0.2061619f, 0.0585292f, -0.0158751f, 0.0036635f, -0.0905686f,
0.3924640f, 0.1691985f, -0.0480354f, 0.0130288f, -0.0030066f, 0.0089078f, -0.0386003f, 0.6271625f,
-0.1780511f, 0.0482933f, -0.0111446f, 0.0349973f, -0.1516549f, 0.5591316f, 0.0305291f, -0.0082805f,
0.0019109f, 0.0019109f, -0.0082805f, 0.0305291f, 0.5591316f, -0.1516549f, 0.0349973f, -0.0111446f,
0.0482933f, -0.1780511f, 0.6271625f, -0.0386003f, 0.0089078f, -0.0030066f, 0.0130288f, -0.0480354f,
0.1691985f, 0.3924640f, -0.0905686f, 0.0036635f, -0.0158751f, 0.0585292f, -0.2061619f, 0.7033362f,
-0.1149711f, 0.0012212f, -0.0052917f, 0.0195097f, -0.0687206f, 0.2344454f, 0.3719327f, -0.0009159f,
0.0039688f, -0.0146323f, 0.0515405f, -0.1758341f, 0.7979735f, 0.8748942f, -0.1721068f, 0.0498959f,
-0.0156577f, 0.0051759f, -0.0015183f, 0.0003136f, 0.2859561f, 0.3933870f, -0.1140478f, 0.0357889f,
-0.0118306f, 0.0034705f, -0.0007168f, -0.1582316f, 0.7660421f, -0.1602371f, 0.0502835f, -0.0166220f,
0.0048760f, -0.0010072f, -0.0333551f, 0.1614811f, 0.5716645f, -0.1793921f, 0.0593008f, -0.0173957f,
0.0035932f, 0.0376046f, -0.1820540f, 0.7985786f, -0.0774646f, 0.0256071f, -0.0075117f, 0.0015516f,
0.0007307f, -0.0035377f, 0.0155179f, 0.6876691f, -0.2273195f, 0.0666833f, -0.0137739f, -0.0104493f,
0.0505878f, -0.2219028f, 0.6858636f, 0.0164518f, -0.0048261f, 0.0009969f, 0.0012696f, -0.0061464f,
0.0269613f, -0.0833326f, 0.8178639f, -0.2399172f, 0.0495567f, 0.0026656f, -0.0129047f, 0.0566062f,
-0.1749600f, 0.5587704f, 0.1561008f, -0.0322438f, -0.0006815f, 0.0032996f, -0.0144735f, 0.0447351f,
-0.1428707f, 0.7886099f, -0.1628932f, -0.0007168f, 0.0034705f, -0.0152231f, 0.0470521f, -0.1502703f,
0.4469841f, 0.2748853f, 0.0003136f, -0.0015183f, 0.0066601f, -0.0205853f, 0.0657433f, -0.1955555f,
0.8797377f, 0.9262632f, -0.2411474f, 0.0556399f, -0.0149728f, 0.0047774f, -0.0011027f, 0.0002986f,
-0.0000614f, 0.1966316f, 0.6430596f, -0.1483731f, 0.0399274f, -0.0127397f, 0.0029406f, -0.0007962f,
0.0001636f, -0.1669505f, 0.8124927f, 0.0469259f, -0.0126278f, 0.0040292f, -0.0009300f, 0.0002518f,
-0.0000517f, 0.0370907f, -0.1805080f, 0.9206194f, -0.2477401f, 0.0790471f, -0.0182455f, 0.0049403f,
-0.0010151f, 0.0159203f, -0.0774786f, 0.2861435f, 0.5380877f, -0.1716891f, 0.0396290f, -0.0107303f,
0.0022049f, -0.0112696f, 0.0548455f, -0.2025551f, 0.8775509f, -0.0836501f, 0.0193080f, -0.0052280f,
0.0010742f, 0.0010742f, -0.0052280f, 0.0193080f, -0.0836501f, 0.8775509f, -0.2025551f, 0.0548455f,
-0.0112696f, 0.0022049f, -0.0107303f, 0.0396290f, -0.1716891f, 0.5380877f, 0.2861435f, -0.0774786f,
0.0159203f, -0.0010151f, 0.0049403f, -0.0182455f, 0.0790471f, -0.2477401f, 0.9206194f, -0.1805080f,
0.0370907f, -0.0000517f, 0.0002518f, -0.0009300f, 0.0040292f, -0.0126278f, 0.0469259f, 0.8124927f,
-0.1669505f, 0.0001636f, -0.0007962f, 0.0029406f, -0.0127397f, 0.0399274f, -0.1483731f, 0.6430596f,
0.1966316f, -0.0000614f, 0.0002986f, -0.0011027f, 0.0047774f, -0.0149728f, 0.0556399f, -0.2411474f,
0.9262632f, 0.9815653f, -0.2288699f, 0.0695333f, -0.0149811f, 0.0048350f, -0.0011356f, 0.0003056f,
-0.0000889f, 0.0000157f, 0.0737389f, 0.9154796f, -0.2781330f, 0.0599244f, -0.0193400f, 0.0045422f,
-0.0012223f, 0.0003556f, -0.0000628f, -0.0983185f, 0.5571383f, 0.3708440f, -0.0798992f, 0.0257866f,
-0.0060563f, 0.0016297f, -0.0004742f, 0.0000837f, 0.0536868f, -0.3042253f, 1.0456146f, -0.0337669f,
0.0108979f, -0.0025595f, 0.0006888f, -0.0002004f, 0.0000354f, -0.0096989f, 0.0549607f, -0.1888986f,
0.9982641f, -0.3221796f, 0.0756679f, -0.0203621f, 0.0059244f, -0.0010455f, -0.0025960f, 0.0147109f,
-0.0505609f, 0.1878900f, 0.8000000f, -0.1878900f, 0.0505609f, -0.0147109f, 0.0025960f, 0.0025960f,
-0.0147109f, 0.0505609f, -0.1878900f, 0.8000000f, 0.1878900f, -0.0505609f, 0.0147109f, -0.0025960f,
-0.0010455f, 0.0059244f, -0.0203621f, 0.0756679f, -0.3221796f, 0.9982641f, -0.1888986f, 0.0549607f,
-0.0096989f, 0.0000354f, -0.0002004f, 0.0006888f, -0.0025595f, 0.0108979f, -0.0337669f, 1.0456146f,
-0.3042253f, 0.0536868f, 0.0000837f, -0.0004742f, 0.0016297f, -0.0060563f, 0.0257866f, -0.0798992f,
0.3708440f, 0.5571383f, -0.0983185f, -0.0000628f, 0.0003556f, -0.0012223f, 0.0045422f, -0.0193400f,
0.0599244f, -0.2781330f, 0.9154796f, 0.0737389f, 0.0000157f, -0.0000889f, 0.0003056f, -0.0011356f,
0.0048350f, -0.0149811f, 0.0695333f, -0.2288699f, 0.9815653f, 0.9974228f, -0.2132571f, 0.0803035f,
-0.0276114f, 0.0056224f, -0.0012128f, 0.0003452f, -0.0001335f, 0.0000402f, -0.0000059f, 0.0137454f,
1.1373711f, -0.4282856f, 0.1472608f, -0.0299864f, 0.0064685f, -0.0018411f, 0.0007119f, -0.0002144f,
0.0000315f, -0.0297817f, 0.2023627f, 0.9279522f, -0.3190651f, 0.0649705f, -0.0140151f, 0.0039891f,
-0.0015424f, 0.0004644f, -0.0000684f, 0.0330908f, -0.2248475f, 0.7467198f, 0.3545168f, -0.0721895f,
0.0155723f, -0.0044323f, 0.0017138f, -0.0005161f, 0.0000759f, -0.0193029f, 0.1311610f, -0.4355866f,
1.1265318f, 0.0421105f, -0.0090838f, 0.0025855f, -0.0009997f, 0.0003010f, -0.0000443f, 0.0051952f,
-0.0353010f, 0.1172349f, -0.3031976f, 1.0652362f, -0.1495209f, 0.0425580f, -0.0164556f, 0.0049550f,
-0.0007292f, -0.0003584f, 0.0024355f, -0.0080883f, 0.0209182f, -0.0734929f, 1.1395742f, -0.3243563f,
0.1254161f, -0.0377645f, 0.0055578f, -0.0000443f, 0.0003010f, -0.0009997f, 0.0025855f, -0.0090838f,
0.0488706f, 1.1246078f, -0.4348426f, 0.1309370f, -0.0192700f, 0.0000759f, -0.0005161f, 0.0017138f,
-0.0044323f, 0.0155723f, -0.0837781f, 0.3578153f, 0.7454444f, -0.2244634f, 0.0330342f, -0.0000684f,
0.0004644f, -0.0015424f, 0.0039891f, -0.0140151f, 0.0754003f, -0.3220338f, 0.9291000f, 0.2020171f,
-0.0297308f, 0.0000315f, -0.0002144f, 0.0007119f, -0.0018411f, 0.0064685f, -0.0348001f, 0.1486310f,
-0.4288154f, 1.1375306f, 0.0137219f, -0.0000059f, 0.0000402f, -0.0001335f, 0.0003452f, -0.0012128f,
0.0065250f, -0.0278683f, 0.0804029f, -0.2132870f, 0.9974272f, 0.9999995f, -0.0666571f, 0.0153275f,
-0.0049158f, 0.0024855f, -0.0011966f, 0.0005915f, -0.0002124f, 0.0000571f, -0.0000096f, 0.0000006f,
0.0000089f, 1.0665138f, -0.2452399f, 0.0786531f, -0.0397674f, 0.0191453f, -0.0094634f, 0.0033982f,
-0.0009139f, 0.0001529f, -0.0000089f, -0.0000446f, 0.0007645f, 1.2261996f, -0.3932655f, 0.1988369f,
-0.0957265f, 0.0473169f, -0.0169909f, 0.0045696f, -0.0007645f, 0.0000446f, 0.0001450f, -0.0024845f,
0.0148512f, 1.2781130f, -0.6462200f, 0.3111111f, -0.1537800f, 0.0552203f, -0.0148512f, 0.0024845f,
-0.0001450f, -0.0002900f, 0.0049690f, -0.0297024f, 0.1104406f, 1.2924401f, -0.6222222f, 0.3075600f,
-0.1104406f, 0.0297024f, -0.0049690f, 0.0002900f, 0.0004143f, -0.0070986f, 0.0424320f, -0.1577723f,
0.4393714f, 0.8888889f, -0.4393714f, 0.1577723f, -0.0424320f, 0.0070986f, -0.0004143f, -0.0004143f,
0.0070986f, -0.0424320f, 0.1577723f, -0.4393714f, 0.8888889f, 0.4393714f, -0.1577723f, 0.0424320f,
-0.0070986f, 0.0004143f, 0.0002900f, -0.0049690f, 0.0297024f, -0.1104406f, 0.3075600f, -0.6222222f,
1.2924401f, 0.1104406f, -0.0297024f, 0.0049690f, -0.0002900f, -0.0001450f, 0.0024845f, -0.0148512f,
0.0552203f, -0.1537800f, 0.3111111f, -0.6462200f, 1.2781130f, 0.0148512f, -0.0024845f, 0.0001450f,
0.0000446f, -0.0007645f, 0.0045696f, -0.0169909f, 0.0473169f, -0.0957265f, 0.1988369f, -0.3932655f,
1.2261996f, 0.0007645f, -0.0000446f, -0.0000089f, 0.0001529f, -0.0009139f, 0.0033982f, -0.0094634f,
0.0191453f, -0.0397674f, 0.0786531f, -0.2452399f, 1.0665138f, 0.0000089f, 0.0000006f, -0.0000096f,
0.0000571f, -0.0002124f, 0.0005915f, -0.0011966f, 0.0024855f, -0.0049158f, 0.0153275f, -0.0666571f,
0.9999995f

View File

@@ -0,0 +1,441 @@
// basisu_bc15_spmd.cpp -- standalone opaque 4-color BC1 encoder (see basisu_bc15_spmd.h).
//
// This TU holds the public API, the solid-color omatch tables, and the scalar reference encoder. The SSE4.1
// cppspmd kernel lives in basisu_bc15_spmd_kernels.inl, compiled by basisu_bc15_spmd_sse.cpp; we only declare +
// call its wrapper here (gated on g_cpu_supports_sse41), with the scalar path as the fallback.
#include "basisu_bc15_spmd.h"
#include "basisu_enc.h"
#include <math.h>
namespace basisu
{
extern bool g_cpu_supports_sse41; // set by detect_sse41() in basisu_encoder_init()
namespace bc_spmd
{
// Baked-in benchmarked defaults (not exposed as parameters by design).
static const int kLsRounds = 2; // least-squares alternation rounds
static const int kAvgVar = 16; // try the avg-solid 2nd seed when (rng_r+rng_g+rng_b) < kAvgVar
#if BASISU_SUPPORT_SSE
// Defined in basisu_bc15_spmd_kernels.inl (the SSE4.1 TU). Encodes up to 4 blocks at once, writing num_write
// of them directly at pOut + j*out_stride (no temp buffer).
void encode_bc1_blocks_4_sse41(const color_rgba* pBlocks, uint8_t* pOut,
int* om5lo, int* om5hi, int* om6lo, int* om6hi, int ls_rounds, int avgvar,
uint32_t out_stride, uint32_t num_write);
// BC4: encodes up to 4 single-channel blocks at once, writing num_write of them at pOut + j*out_stride.
// do_ls!=0 adds the 1-D least-squares refit. src_stride = byte stride between texel values in pBlocks (1 or 4).
void encode_bc4_blocks_4_sse41(const uint8_t* pBlocks, uint8_t* pOut, uint32_t out_stride, uint32_t num_write, uint32_t do_ls, uint32_t src_stride);
#endif
// ---- solid-color omatch tables (stb_dxt/basisu 3% span penalty), as int[256] for the kernel's gather ----
static int s_om5_lo[256], s_om5_hi[256], s_om6_lo[256], s_om6_hi[256];
static void build_omatch_tables()
{
for (int bits = 5; bits <= 6; bits++)
{
const int levels = 1 << bits;
int* lo = (bits == 5) ? s_om5_lo : s_om6_lo;
int* hi = (bits == 5) ? s_om5_hi : s_om6_hi;
for (int target = 0; target < 256; target++)
{
int best = 0x7FFFFFFF, bc0 = 0, bc1 = 0;
for (int c0 = 0; c0 < levels; c0++)
{
const int e0 = (bits == 5) ? ((c0 << 3) | (c0 >> 2)) : ((c0 << 2) | (c0 >> 4));
for (int c1 = 0; c1 < levels; c1++) // FULL cross product (matches encode_bc1's prepare_bc1_single_color_table)
{
const int e1 = (bits == 5) ? ((c1 << 3) | (c1 >> 2)) : ((c1 << 2) | (c1 >> 4));
const int interp = (2 * e0 + e1) / 3; // selector-2 (2/3 toward c0), truncated
int err = (interp > target) ? (interp - target) : (target - interp);
err += ((e0 > e1 ? e0 - e1 : e1 - e0) * 3) / 100; // 3% linear span penalty (abs: e0 may be < e1 now)
if (err < best) { best = err; bc0 = c0; bc1 = c1; }
}
}
lo[target] = bc0; hi[target] = bc1;
}
}
}
// Thread-safe one-time table init: C++11 guarantees a function-local static's initializer runs exactly once,
// race-free, even under concurrent first calls (no double-checked-locking bug). init() just forces it.
static inline void ensure_init() { static const bool s_built = (build_omatch_tables(), true); (void)s_built; }
void init() { ensure_init(); }
// ---- scalar reference encoder (the oracle) -- identical algorithm/op-order to the SPMD kernel ----
static inline void unpack565(uint32_t c, int& r, int& g, int& b)
{
r = (c >> 11) & 31; g = (c >> 5) & 63; b = c & 31;
r = (r << 3) | (r >> 2); g = (g << 2) | (g >> 4); b = (b << 3) | (b >> 2);
}
static inline uint32_t f_to_565(float r, float g, float b)
{
const float cr = (r < 0.0f) ? 0.0f : (r > 255.0f ? 255.0f : r);
const float cg = (g < 0.0f) ? 0.0f : (g > 255.0f ? 255.0f : g);
const float cb = (b < 0.0f) ? 0.0f : (b > 255.0f ? 255.0f : b);
return ((int)rintf(cr * (31.0f / 255.0f)) << 11) | ((int)rintf(cg * (63.0f / 255.0f)) << 5) | (int)rintf(cb * (31.0f / 255.0f));
}
static inline int thresh_sel(const int* pr, const int* pg, const int* pb, int r, int g, int b)
{
const int ar = pr[1] - pr[0], ag = pg[1] - pg[0], ab = pb[1] - pb[0];
const int dp0 = pr[0] * ar + pg[0] * ag + pb[0] * ab;
const int dp2 = pr[2] * ar + pg[2] * ag + pb[2] * ab;
const int dp3 = pr[3] * ar + pg[3] * ag + pb[3] * ab;
const int dp1 = pr[1] * ar + pg[1] * ag + pb[1] * ab;
const int t0 = dp0 + dp2, t1 = dp2 + dp3, t2 = dp3 + dp1;
const int d = r * (ar + ar) + g * (ag + ag) + b * (ab + ab);
return (d <= t0) ? 0 : (d < t1) ? 2 : (d < t2) ? 3 : 1;
}
static void ls_round(const color_rgba* blk, uint32_t& c0, uint32_t& c1)
{
int r0, g0, b0, r1, g1, b1;
unpack565(c0, r0, g0, b0); unpack565(c1, r1, g1, b1);
const int pr[4] = { r0, r1, (2 * r0 + r1) / 3, (r0 + 2 * r1) / 3 };
const int pg[4] = { g0, g1, (2 * g0 + g1) / 3, (g0 + 2 * g1) / 3 };
const int pb[4] = { b0, b1, (2 * b0 + b1) / 3, (b0 + 2 * b1) / 3 };
float m00 = 0, m01 = 0, m11 = 0, a0r = 0, a0g = 0, a0b = 0, a1r = 0, a1g = 0, a1b = 0;
for (uint32_t i = 0; i < 16; i++)
{
const int sel = thresh_sel(pr, pg, pb, blk[i].r, blk[i].g, blk[i].b);
const float tb = (sel == 0) ? 0.0f : (sel == 1) ? 1.0f : (sel == 2) ? (1.0f / 3.0f) : (2.0f / 3.0f);
const float ta = 1.0f - tb;
m00 += ta * ta; m01 += ta * tb; m11 += tb * tb;
const float cr = (float)blk[i].r, cg = (float)blk[i].g, cb = (float)blk[i].b;
a0r += ta * cr; a0g += ta * cg; a0b += ta * cb;
a1r += tb * cr; a1g += tb * cg; a1b += tb * cb;
}
const float det = m00 * m11 - m01 * m01;
const float inv = 1.0f / basisu::maximum(det, 1e-6f);
const float e0r = (a0r * m11 - a1r * m01) * inv, e0g = (a0g * m11 - a1g * m01) * inv, e0b = (a0b * m11 - a1b * m01) * inv;
const float e1r = (a1r * m00 - a0r * m01) * inv, e1g = (a1g * m00 - a0g * m01) * inv, e1b = (a1b * m00 - a0b * m01) * inv;
uint32_t nc0 = f_to_565(e0r, e0g, e0b), nc1 = f_to_565(e1r, e1g, e1b);
{ const uint32_t hi = basisu::maximum(nc0, nc1), lo = basisu::minimum(nc0, nc1); nc0 = hi; nc1 = lo; }
if (det > 1e-6f) { c0 = nc0; c1 = nc1; }
}
static int eval_error(const color_rgba* blk, uint32_t c0, uint32_t c1)
{
int r0, g0, b0, r1, g1, b1;
unpack565(c0, r0, g0, b0); unpack565(c1, r1, g1, b1);
const int pr[4] = { r0, r1, (2 * r0 + r1) / 3, (r0 + 2 * r1) / 3 };
const int pg[4] = { g0, g1, (2 * g0 + g1) / 3, (g0 + 2 * g1) / 3 };
const int pb[4] = { b0, b1, (2 * b0 + b1) / 3, (b0 + 2 * b1) / 3 };
int err = 0;
for (uint32_t i = 0; i < 16; i++)
{
const int sel = thresh_sel(pr, pg, pb, blk[i].r, blk[i].g, blk[i].b);
const int dr = blk[i].r - pr[sel], dg = blk[i].g - pg[sel], db = blk[i].b - pb[sel];
err += dr * dr + dg * dg + db * db;
}
return err;
}
static void encode_block_scalar(const color_rgba* blk, uint8_t out[8])
{
int mnr = 255, mng = 255, mnb = 255, mxr = 0, mxg = 0, mxb = 0, sum_r = 0, sum_g = 0, sum_b = 0;
int sum_rr = 0, sum_gg = 0, sum_bb = 0, sum_rg = 0, sum_rb = 0, sum_gb = 0;
for (uint32_t i = 0; i < 16; i++)
{
const int r = blk[i].r, g = blk[i].g, b = blk[i].b;
mnr = basisu::minimum<int>(mnr, r); mng = basisu::minimum<int>(mng, g); mnb = basisu::minimum<int>(mnb, b);
mxr = basisu::maximum<int>(mxr, r); mxg = basisu::maximum<int>(mxg, g); mxb = basisu::maximum<int>(mxb, b);
sum_r += r; sum_g += g; sum_b += b;
sum_rr += r * r; sum_gg += g * g; sum_bb += b * b;
sum_rg += r * g; sum_rb += r * b; sum_gb += g * b;
}
// Solid block: clone of basist::encode_bc1_solid_block -- omatch endpoints from the (full-cross-product)
// single-color tables, then the same swap + degenerate handling that GUARANTEES a 4-color block
// (color0 > color1), never 3-color (required so this is safe for BC3/BC5 color blocks).
if ((mnr == mxr) && (mng == mxg) && (mnb == mxb))
{
const int ar = (sum_r + 8) >> 4, ag = (sum_g + 8) >> 4, ab = (sum_b + 8) >> 4;
uint32_t max16 = (s_om5_lo[ar] << 11) | (s_om6_lo[ag] << 5) | s_om5_lo[ab]; // lo[] = 2x-weighted endpoint (encode_bc1 m_hi)
uint32_t min16 = (s_om5_hi[ar] << 11) | (s_om6_hi[ag] << 5) | s_om5_hi[ab]; // hi[] = 1x endpoint (encode_bc1 m_lo)
uint32_t mask = 0xAA;
if (min16 == max16)
{
mask = 0; // selector 0 = color0 directly; force max16 > min16 so the block stays 4-color
if (min16 > 0) min16--;
else { max16 = 1; min16 = 0; mask = 0x55; } // l == h == 0
}
if (max16 < min16) { const uint32_t t = max16; max16 = min16; min16 = t; mask ^= 0x55; } // selector 2<->3 on swap
out[0] = (uint8_t)max16; out[1] = (uint8_t)(max16 >> 8); out[2] = (uint8_t)min16; out[3] = (uint8_t)(min16 >> 8);
out[4] = out[5] = out[6] = out[7] = (uint8_t)mask;
return;
}
// PCA seed (METHOD 2, on-axis, sqrt-free).
const float cxx = (float)(16 * sum_rr - sum_r * sum_r), cxy = (float)(16 * sum_rg - sum_r * sum_g), cxz = (float)(16 * sum_rb - sum_r * sum_b);
const float cyy = (float)(16 * sum_gg - sum_g * sum_g), cyz = (float)(16 * sum_gb - sum_g * sum_b), czz = (float)(16 * sum_bb - sum_b * sum_b);
float ax = (float)(mxr - mnr), ay = (float)(mxg - mng), az = (float)(mxb - mnb);
for (int it = 0; it < 4; it++)
{
const float nr = ax * cxx + ay * cxy + az * cxz;
const float ng = ax * cxy + ay * cyy + az * cyz;
const float nb = ax * cxz + ay * cyz + az * czz;
ax = nr; ay = ng; az = nb;
}
const float kk = basisu::maximum(basisu::maximum(fabsf(ax), fabsf(ay)), fabsf(az));
const float mm = 1024.0f / basisu::maximum(kk, 1e-3f);
ax *= mm; ay *= mm; az *= mm;
const float inv_len2 = 1.0f / (ax * ax + ay * ay + az * az + 0.0000125f);
const float meanx = (float)sum_r * (1.0f / 16.0f), meany = (float)sum_g * (1.0f / 16.0f), meanz = (float)sum_b * (1.0f / 16.0f);
float minp = 1e30f, maxp = -1e30f;
for (uint32_t i = 0; i < 16; i++)
{
const float prj = ((float)blk[i].r - meanx) * ax + ((float)blk[i].g - meany) * ay + ((float)blk[i].b - meanz) * az;
minp = basisu::minimum(minp, prj); maxp = basisu::maximum(maxp, prj);
}
const float tlo = minp * inv_len2, thi = maxp * inv_len2;
uint32_t color0 = f_to_565(meanx + tlo * ax, meany + tlo * ay, meanz + tlo * az);
uint32_t color1 = f_to_565(meanx + thi * ax, meany + thi * ay, meanz + thi * az);
{ const uint32_t hi = basisu::maximum(color0, color1), lo = basisu::minimum(color0, color1); color0 = hi; color1 = lo; }
if (color0 == color1) { if (color1 != 0) color1--; else color0++; }
for (int r = 0; r < kLsRounds; r++) ls_round(blk, color0, color1);
if (color0 == color1) { if (color1 != 0) color1--; else color0++; }
// Low-variance avg-solid 2nd seed (keep-better).
const int var_proxy = (mxr - mnr) + (mxg - mng) + (mxb - mnb);
if (var_proxy < kAvgVar)
{
const int ar = (sum_r + 8) >> 4, ag = (sum_g + 8) >> 4, ab = (sum_b + 8) >> 4;
uint32_t c0b = (s_om5_lo[ar] << 11) | (s_om6_lo[ag] << 5) | s_om5_lo[ab];
uint32_t c1b = (s_om5_hi[ar] << 11) | (s_om6_hi[ag] << 5) | s_om5_hi[ab];
{ const uint32_t hi = basisu::maximum(c0b, c1b), lo = basisu::minimum(c0b, c1b); c0b = hi; c1b = lo; }
if (c0b == c1b) { if (c1b != 0) c1b--; else c0b++; }
ls_round(blk, c0b, c1b);
if (c0b == c1b) { if (c1b != 0) c1b--; else c0b++; }
if (eval_error(blk, c0b, c1b) < eval_error(blk, color0, color1)) { color0 = c0b; color1 = c1b; }
}
int r0, g0, b0, r1, g1, b1;
unpack565(color0, r0, g0, b0); unpack565(color1, r1, g1, b1);
const int pr[4] = { r0, r1, (2 * r0 + r1) / 3, (r0 + 2 * r1) / 3 };
const int pg[4] = { g0, g1, (2 * g0 + g1) / 3, (g0 + 2 * g1) / 3 };
const int pb[4] = { b0, b1, (2 * b0 + b1) / 3, (b0 + 2 * b1) / 3 };
uint32_t sw = 0;
for (uint32_t i = 0; i < 16; i++)
sw |= ((uint32_t)thresh_sel(pr, pg, pb, blk[i].r, blk[i].g, blk[i].b)) << (i * 2);
out[0] = (uint8_t)color0; out[1] = (uint8_t)(color0 >> 8);
out[2] = (uint8_t)color1; out[3] = (uint8_t)(color1 >> 8);
out[4] = (uint8_t)sw; out[5] = (uint8_t)(sw >> 8); out[6] = (uint8_t)(sw >> 16); out[7] = (uint8_t)(sw >> 24);
}
// ---- public API ----
void encode_bc1_scalar(void* pBlocks, const color_rgba* pSrc_pixels, uint32_t num_blocks, uint32_t block_stride)
{
ensure_init();
uint8_t* pOut = (uint8_t*)pBlocks;
const uint32_t out_stride = 8 * block_stride;
for (uint32_t b = 0; b < num_blocks; b++)
encode_block_scalar(pSrc_pixels + b * 16, pOut + b * out_stride);
}
void encode_bc1_spmd(void* pBlocks, const color_rgba* pSrc_pixels, uint32_t num_blocks, uint32_t block_stride)
{
ensure_init();
uint8_t* pOut = (uint8_t*)pBlocks;
(void)pOut;
const uint32_t out_stride = 8 * block_stride;
(void)out_stride;
#if BASISU_SUPPORT_SSE
if (g_cpu_supports_sse41)
{
// Full groups of 4: the kernel writes all 4 blocks directly at the (possibly strided) output.
uint32_t base = 0;
for (; base + 4 <= num_blocks; base += 4)
encode_bc1_blocks_4_sse41(pSrc_pixels + base * 16, pOut + base * out_stride, s_om5_lo, s_om5_hi, s_om6_lo, s_om6_hi, kLsRounds, kAvgVar, out_stride, 4);
// Final partial group: pad the input to 4, but write only the n valid blocks (still direct, no temp).
const uint32_t n = num_blocks - base; // 0..3
if (n)
{
color_rgba grp[64];
for (uint32_t gi = 0; gi < 4; gi++)
memcpy(grp + gi * 16, pSrc_pixels + (base + ((gi < n) ? gi : (n - 1))) * 16, 16 * sizeof(color_rgba));
encode_bc1_blocks_4_sse41(grp, pOut + base * out_stride, s_om5_lo, s_om5_hi, s_om6_lo, s_om6_hi, kLsRounds, kAvgVar, out_stride, n);
}
return;
}
#endif
encode_bc1_scalar(pBlocks, pSrc_pixels, num_blocks, block_stride); // scalar fallback (no SSE4.1)
}
// ================================ BC4 (RGTC1, single channel) ================================
// Scalar reference (oracle) for one BC4 block -- clone of basist::encode_bc4 (raw bbox min/max, exact
// nearest-of-8 integer-threshold selector), but the solid case is forced to 8-value mode (red0 > red1) so
// every block is 8-color. v = 16 contiguous channel bytes (texel i = y*4+x). out = 8 bytes.
// Compute the nearest-of-8 selectors + the resulting decoded SSE for an 8-value endpoint pair (r0 > r1).
// Optionally accumulates the 1-D least-squares moments (rank = interpolation weight, 0..7).
static int bc4_eval(const uint8_t* v, int r0, int r1, uint64_t& sel_out, int* pSw, int* pSww, int* pSv, int* pSwv, uint32_t src_stride)
{
const int delta = r0 - r1;
const int t0 = delta * 13, t1 = delta * 11, t2 = delta * 9, t3 = delta * 7, t4 = delta * 5, t5 = delta * 3, t6 = delta;
const int bias = 4 - r1 * 14;
uint64_t sel = 0; int sse = 0, Sw = 0, Sww = 0, Sv = 0, Swv = 0;
for (int i = 0; i < 16; i++)
{
const int val = v[i * src_stride];
const int sv = val * 14 + bias;
const int rank = (sv >= t0) + (sv >= t1) + (sv >= t2) + (sv >= t3) + (sv >= t4) + (sv >= t5) + (sv >= t6);
const int idx = (rank == 7) ? 0 : (rank == 0) ? 1 : (8 - rank);
sel |= (uint64_t)idx << (i * 3);
const int recon = (rank * r0 + (7 - rank) * r1) / 7; // == decoded value for this selector (matches unpack_bc4)
const int d = val - recon; sse += d * d;
Sw += rank; Sww += rank * rank; Sv += val; Swv += rank * val;
}
sel_out = sel;
if (pSw) { *pSw = Sw; *pSww = Sww; *pSv = Sv; *pSwv = Swv; }
return sse;
}
static void encode_bc4_block(const uint8_t* v, uint8_t out[8], bool do_ls, uint32_t src_stride)
{
int mn = v[0], mx = v[0];
for (int i = 1; i < 16; i++) { const int x = v[i * src_stride]; mn = basisu::minimum(mn, x); mx = basisu::maximum(mx, x); }
// Solid -> force an 8-value block (red0 > red1) that still reconstructs the exact value.
if (mn == mx)
{
int r0, r1, idx;
if (mx > 0) { r0 = mx; r1 = mx - 1; idx = 0; } // all selectors index 0 = red0 = value
else { r0 = 1; r1 = 0; idx = 1; } // value 0: all selectors index 1 = red1 = 0
out[0] = (uint8_t)r0; out[1] = (uint8_t)r1;
uint64_t sel = 0; for (int i = 0; i < 16; i++) sel |= (uint64_t)idx << (i * 3);
for (int i = 0; i < 6; i++) out[2 + i] = (uint8_t)(sel >> (i * 8));
return;
}
// Candidate A: raw bbox endpoints (== basist::encode_bc4) + (for HQ) its LS moments.
int Sw, Sww, Sv, Swv; uint64_t sel;
int r0 = mx, r1 = mn;
int sse = bc4_eval(v, r0, r1, sel, &Sw, &Sww, &Sv, &Swv, src_stride);
// Candidate B (HQ only): one 1-D least-squares endpoint refit (value ~= a + (rank/7)*b), adopted only if
// it lowers SSE -- monotonic, can never regress below bbox/encode_bc4. Always 8-value (require nr0 > nr1).
if (do_ls)
{
const int detI = 16 * Sww - Sw * Sw;
if (detI != 0)
{
const float b = 7.0f * (float)(16 * Swv - Sw * Sv) / (float)detI; // slope = red0 - red1
const float bsw7 = b * (float)Sw * (1.0f / 7.0f);
const float a = ((float)Sv - bsw7) / 16.0f; // intercept = red1
const int nr1 = basisu::clamp<int>((int)rintf(a), 0, 255);
const int nr0 = basisu::clamp<int>((int)rintf(a + b), 0, 255);
if (nr0 > nr1)
{
uint64_t selB;
const int sseB = bc4_eval(v, nr0, nr1, selB, nullptr, nullptr, nullptr, nullptr, src_stride);
if (sseB < sse) { r0 = nr0; r1 = nr1; sel = selB; sse = sseB; }
}
}
}
out[0] = (uint8_t)r0; out[1] = (uint8_t)r1;
for (int i = 0; i < 6; i++) out[2 + i] = (uint8_t)(sel >> (i * 8));
}
void encode_bc4_scalar(void* pBlocks, const uint8_t* pSrc_pixels, uint32_t num_blocks, uint32_t block_stride, bool high_quality, uint32_t src_stride)
{
uint8_t* pOut = (uint8_t*)pBlocks;
const uint32_t out_stride = 8 * block_stride;
const uint32_t in_block = 16 * src_stride; // bytes between consecutive blocks in the source
for (uint32_t b = 0; b < num_blocks; b++)
encode_bc4_block(pSrc_pixels + (size_t)b * in_block, pOut + b * out_stride, high_quality, src_stride);
}
void encode_bc4_spmd(void* pBlocks, const uint8_t* pSrc_pixels, uint32_t num_blocks, uint32_t block_stride, bool high_quality, uint32_t src_stride)
{
uint8_t* pOut = (uint8_t*)pBlocks;
(void)pOut;
const uint32_t out_stride = 8 * block_stride;
(void)out_stride;
const uint32_t in_block = 16 * src_stride;
(void)in_block;
#if BASISU_SUPPORT_SSE
if (g_cpu_supports_sse41)
{
const uint32_t do_ls = high_quality ? 1 : 0;
// Full groups read straight from the (possibly strided) source -- no copy.
uint32_t base = 0;
for (; base + 4 <= num_blocks; base += 4)
encode_bc4_blocks_4_sse41(pSrc_pixels + (size_t)base * in_block, pOut + base * out_stride, out_stride, 4, do_ls, src_stride);
// Tail (<4): gather the channel into a tightly-packed 64-byte stack buffer (pad with last block), then
// encode with src_stride=1. Only the tail copies; the bulk above is zero-copy.
const uint32_t n = num_blocks - base; // 0..3
if (n)
{
uint8_t grp[64];
for (uint32_t gi = 0; gi < 4; gi++)
{
const uint8_t* sp = pSrc_pixels + (size_t)(base + ((gi < n) ? gi : (n - 1))) * in_block;
for (uint32_t t = 0; t < 16; t++) grp[gi * 16 + t] = sp[t * src_stride];
}
encode_bc4_blocks_4_sse41(grp, pOut + base * out_stride, out_stride, n, do_ls, 1);
}
return;
}
#endif
encode_bc4_scalar(pBlocks, pSrc_pixels, num_blocks, block_stride, high_quality, src_stride); // scalar fallback
}
// ================================ High-level RGBA format helpers ================================
// NO allocations, NO channel-extraction copies: the BC4 path reads the requested channel straight out of the
// RGBA pixels via src_stride=4 (the channel pointer is &pPixels->r/g/b/a). encode_bc4_* handles the <4 tail
// without overrunning the (possibly strided) output.
void encode_bc1(void* pBlocks, const color_rgba* pPixels, uint32_t num_blocks, bool use_spmd)
{
if (use_spmd) encode_bc1_spmd(pBlocks, pPixels, num_blocks, 1);
else encode_bc1_scalar(pBlocks, pPixels, num_blocks, 1);
}
void encode_bc4(void* pBlocks, const color_rgba* pPixels, uint32_t num_blocks, bool use_spmd, bool high_quality)
{
const uint8_t* pR = &pPixels[0].r; // R channel, stride 4, contiguous output
if (use_spmd) encode_bc4_spmd(pBlocks, pR, num_blocks, 1, high_quality, 4);
else encode_bc4_scalar(pBlocks, pR, num_blocks, 1, high_quality, 4);
}
void encode_bc5(void* pBlocks, const color_rgba* pPixels, uint32_t num_blocks, bool use_spmd, bool high_quality)
{
uint8_t* pOut = (uint8_t*)pBlocks; // [0..7] = BC4 of R, [8..15] = BC4 of G (each block_stride 2)
const uint8_t* pR = &pPixels[0].r, * pG = &pPixels[0].g;
if (use_spmd) { encode_bc4_spmd(pOut, pR, num_blocks, 2, high_quality, 4); encode_bc4_spmd(pOut + 8, pG, num_blocks, 2, high_quality, 4); }
else { encode_bc4_scalar(pOut, pR, num_blocks, 2, high_quality, 4); encode_bc4_scalar(pOut + 8, pG, num_blocks, 2, high_quality, 4); }
}
void encode_bc3(void* pBlocks, const color_rgba* pPixels, uint32_t num_blocks, bool use_spmd, bool high_quality)
{
uint8_t* pOut = (uint8_t*)pBlocks; // [0..7] = BC4 alpha block, [8..15] = BC1 color block
const uint8_t* pA = &pPixels[0].a; // A channel, stride 4
if (use_spmd) { encode_bc4_spmd(pOut, pA, num_blocks, 2, high_quality, 4); encode_bc1_spmd(pOut + 8, pPixels, num_blocks, 2); }
else { encode_bc4_scalar(pOut, pA, num_blocks, 2, high_quality, 4); encode_bc1_scalar(pOut + 8, pPixels, num_blocks, 2); }
}
void encode_bc2(void* pBlocks, const color_rgba* pPixels, uint32_t num_blocks, bool use_spmd)
{
uint8_t* pOut = (uint8_t*)pBlocks; // [0..7] = explicit 4-bit alpha (scalar), [8..15] = BC1 color block
for (uint32_t b = 0; b < num_blocks; b++)
{
uint8_t* o = pOut + b * 16;
const color_rgba* p = pPixels + b * 16;
for (uint32_t y = 0; y < 4; y++)
{
uint32_t row = 0; // one little-endian 16-bit word per row; texel x in nibble x
for (uint32_t x = 0; x < 4; x++) { const int a8 = p[y * 4 + x].a; const int a4 = (a8 * 2 + 17) / 34; row |= (uint32_t)a4 << (x * 4); } // round(a8/17)
o[y * 2] = (uint8_t)row; o[y * 2 + 1] = (uint8_t)(row >> 8);
}
}
if (use_spmd) encode_bc1_spmd(pOut + 8, pPixels, num_blocks, 2);
else encode_bc1_scalar(pOut + 8, pPixels, num_blocks, 2);
}
} // namespace bc_spmd
} // namespace basisu

View File

@@ -0,0 +1,73 @@
// basisu_bc15_spmd.h
// Standalone opaque 4-color BC1 (DXT1) encoder for the Basis Universal encoder library.
//
// Two entry points share the SAME algorithm (omatch-solid fast path -> PCA endpoint seed -> 2x integer-threshold
// least-squares -> low-variance avg-solid 2nd seed (keep-better) -> integer-threshold selectors):
// - encode_bc1_scalar : portable scalar reference ("oracle"), no SIMD.
// - encode_bc1_spmd : SSE4.1 cppspmd kernel, 4 blocks per vector; falls back to the scalar path when SSE4.1
// is unavailable (e.g. WASM without SIMD).
// Benchmarked (vs basist::encode_bc1): ~3.4-4.7x faster on photos, ~+0.06 dB average, and it beats encode_bc1
// on low-dynamic-range / gradient blocks. Opaque 4-color only (no 3-color / punchthrough).
#pragma once
#include "basisu_enc.h" // basisu::color_rgba
namespace basisu
{
namespace bc_spmd
{
// Build the solid-color omatch tables. Call once before encoding (idempotent). Not safe to call
// concurrently with itself or the encoders on first use -- call it once at startup (e.g. right after
// basisu_encoder_init(), which also performs the CPU-feature detection encode_bc1_spmd dispatches on).
void init();
// Encode num_blocks opaque BC1 blocks.
// pSrc_pixels : num_blocks * 16 color_rgba, block-contiguous; texel index within a block = y*4 + x.
// pBlocks : output; BC1 block b is written at (uint8_t*)pBlocks + b * 8 * block_stride.
// num_blocks : any count; counts not divisible by 4 are handled internally (no caller padding needed).
// block_stride : output spacing in units of 8-byte BC1 blocks. 1 = contiguous BC1. 2 = write into every
// other 8-byte slot, e.g. the color half of 16-byte BC3/BC5 blocks (pass pBlocks already
// offset to that half). Input layout is unaffected.
// The caller is responsible for extracting the 16 pixels per block from its source image.
void encode_bc1_scalar(void* pBlocks, const color_rgba* pSrc_pixels, uint32_t num_blocks, uint32_t block_stride = 1);
void encode_bc1_spmd(void* pBlocks, const color_rgba* pSrc_pixels, uint32_t num_blocks, uint32_t block_stride = 1);
// Encode num_blocks single-channel BC4 (RGTC1) blocks. Clone of basist::encode_bc4's algorithm: raw bbox
// min/max endpoints + exact nearest-of-8 integer-threshold selector, ALWAYS 8-value mode (red0 > red1) --
// including a forced-8-value solid-block path so we NEVER emit a 6-value block (safe for BC3/BC5). Quality
// matches encode_bc4 (bit-exact on non-solid blocks; identical decode on solid). (Least-squares endpoint
// refinement may be layered on later.)
// pSrc_pixels : num_blocks * 16 uint8_t, block-contiguous; texel index within a block = y*4 + x. The
// caller extracts the 16 single-channel values per block itself (e.g. one channel of RGBA).
// pBlocks : output; BC4 block b is written at (uint8_t*)pBlocks + b * 8 * block_stride.
// num_blocks : any count; the <4 tail is handled internally.
// block_stride : output spacing in units of 8-byte BC4 blocks. 1 = contiguous BC4. 2 = write into every
// other 8-byte slot, e.g. the alpha half of a 16-byte BC3 block or one half of a BC5 pair.
// high_quality : false (default) = fast bbox encoder, matches encode_bc4 quality, ~1.4x faster. true = add a
// 1-D least-squares endpoint refit (keep-best, monotonic): ~+0.8 dB on photos but ~0.6x speed.
// src_stride : BYTE stride between consecutive texel values in pSrc_pixels. 1 (default) = tightly packed
// single channel. 4 = one channel of color_rgba (point pSrc_pixels at the desired channel
// byte, e.g. &rgba[0].g) -- lets callers encode a channel in place with no extraction copy.
void encode_bc4_scalar(void* pBlocks, const uint8_t* pSrc_pixels, uint32_t num_blocks, uint32_t block_stride = 1, bool high_quality = false, uint32_t src_stride = 1);
void encode_bc4_spmd(void* pBlocks, const uint8_t* pSrc_pixels, uint32_t num_blocks, uint32_t block_stride = 1, bool high_quality = false, uint32_t src_stride = 1);
// ---------------- High-level format helpers (RGBA in -> complete GPU blocks out) ----------------
// Convenience wrappers over the low-level encoders. Each takes pPixels = num_blocks * 16 color_rgba,
// block-contiguous (texel index = y*4 + x), and writes complete blocks. num_blocks may be ANY count >= 1
// (not necessarily a multiple of 4); the <4 tail is handled internally with NO output overrun.
// use_spmd : true = SSE4.1 SPMD path (auto-falls back to scalar when SSE4.1 is unavailable); false = scalar.
// high_quality : enables the BC4 channel least-squares refit for the alpha/red/green block(s) (BC3/BC4/BC5).
// Output block sizes / layout (matches the transcoder's bcu unpackers):
// BC1: 8 bytes/block -- RGB, opaque 4-color.
// BC2: 16 bytes/block -- explicit 4-bit alpha [0..7] (scalar) + BC1 color [8..15].
// BC3: 16 bytes/block -- BC4 alpha block [0..7] + BC1 color [8..15].
// BC4: 8 bytes/block -- single channel = R.
// BC5: 16 bytes/block -- BC4 of R [0..7] + BC4 of G [8..15].
void encode_bc1(void* pBlocks, const color_rgba* pPixels, uint32_t num_blocks, bool use_spmd = true);
void encode_bc2(void* pBlocks, const color_rgba* pPixels, uint32_t num_blocks, bool use_spmd = true);
void encode_bc3(void* pBlocks, const color_rgba* pPixels, uint32_t num_blocks, bool use_spmd = true, bool high_quality = false);
void encode_bc4(void* pBlocks, const color_rgba* pPixels, uint32_t num_blocks, bool use_spmd = true, bool high_quality = false);
void encode_bc5(void* pBlocks, const color_rgba* pPixels, uint32_t num_blocks, bool use_spmd = true, bool high_quality = false);
} // namespace bc_spmd
} // namespace basisu

View File

@@ -0,0 +1,428 @@
// basisu_bc15_spmd_kernels.inl -- Do NOT directly include.
//
// SSE4.1 cppspmd kernel for the standalone BC1 encoder. Included by basisu_bc15_spmd_sse.cpp from file scope after
// cppspmd_sse.h + "using namespace CPPSPMD;". LANES = BLOCKS: 4 independent BC1 blocks per vector, one per lane;
// each lane runs the same scalar-looking program on its block's 16 texels. Mirrors encoder/basisu_kernels_imp.h.
namespace bc_spmd_kern
{
// Opaque 4-color BC1 encode of 4 blocks at once (one block per lane). pBlocks = 4 blocks x 16 texels,
// block-contiguous; pOut = 4*8 bytes. om5*/om6* = solid-color omatch tables (int[256]); ls_rounds, avgvar tune.
struct encode_bc1_4color_blocks : spmd_kernel
{
inline vint div3(const vint& x) { return VUINT_SHIFT_RIGHT(x * vint(43691), 17); } // exact floor(x/3), x small
inline vint expand5(const vint& c) { return VINT_SHIFT_LEFT(c, 3) | VUINT_SHIFT_RIGHT(c, 2); }
inline vint expand6(const vint& c) { return VINT_SHIFT_LEFT(c, 2) | VUINT_SHIFT_RIGHT(c, 4); }
// Round per-lane float RGB (0..255) STRAIGHT into 5/6/5-bit space (single round, like encode_bc1's
// *31/255+0.5) -- avoids the double-rounding (float->8bit->565) that biases endpoints inward.
inline vint f_to_565(const vfloat& r, const vfloat& g, const vfloat& b)
{
const vint r5 = vint(round_nearest(min(max(r, vfloat(0.0f)), vfloat(255.0f)) * vfloat(31.0f / 255.0f)));
const vint g6 = vint(round_nearest(min(max(g, vfloat(0.0f)), vfloat(255.0f)) * vfloat(63.0f / 255.0f)));
const vint b5 = vint(round_nearest(min(max(b, vfloat(0.0f)), vfloat(255.0f)) * vfloat(31.0f / 255.0f)));
return VINT_SHIFT_LEFT(r5, 11) | VINT_SHIFT_LEFT(g6, 5) | b5;
}
// Load texel i of all 4 blocks into per-channel vints (lane j = block j's texel i).
inline void load_texel(const int32_t* p, int i, vint& r, vint& g, vint& b)
{
vint rgba; rgba.m_value = _mm_setr_epi32(p[i], p[16 + i], p[32 + i], p[48 + i]);
r = rgba & vint(0xFF);
g = VINT_SHIFT_RIGHT(rgba, 8) & vint(0xFF);
b = VINT_SHIFT_RIGHT(rgba, 16) & vint(0xFF);
// Opaque 4-color encoder: alpha is ignored entirely.
}
// Integer-threshold selector (encode_bc1's bc1_find_sels scheme): 1 dot + 3 compares/texel. The asymmetric
// (d<=t0) biases extreme texels onto the OUTER selectors so the LS step reaches wider endpoints.
// Palette order is [pal0,pal2,pal3,pal1] along the c0->c1 axis; lut4={1,3,2,0} via the ternary cascade.
struct thresholds { vint t0, t1, t2, ar2, ag2, ab2; };
inline thresholds make_thresholds(
const vint& pr0, const vint& pg0, const vint& pb0, const vint& pr1, const vint& pg1, const vint& pb1,
const vint& pr2, const vint& pg2, const vint& pb2, const vint& pr3, const vint& pg3, const vint& pb3)
{
const vint ar = pr1 - pr0, ag = pg1 - pg0, ab = pb1 - pb0;
const vint dp0 = pr0 * ar + pg0 * ag + pb0 * ab;
const vint dp2 = pr2 * ar + pg2 * ag + pb2 * ab;
const vint dp3 = pr3 * ar + pg3 * ag + pb3 * ab;
const vint dp1 = pr1 * ar + pg1 * ag + pb1 * ab;
thresholds T;
T.t0 = dp0 + dp2; T.t1 = dp2 + dp3; T.t2 = dp3 + dp1;
T.ar2 = ar + ar; T.ag2 = ag + ag; T.ab2 = ab + ab;
return T;
}
inline vint thresh_sel(const thresholds& T, const vint& r, const vint& g, const vint& b)
{
const vint d = r * T.ar2 + g * T.ag2 + b * T.ab2;
return spmd_ternaryi(d <= T.t0, 0, spmd_ternaryi(d < T.t1, 2, spmd_ternaryi(d < T.t2, 3, 1)));
}
// One least-squares round (per lane): assign integer-threshold selectors for the current endpoints, refit
// the endpoint line via the 2x2 normal equations (Cramer), adopt only where well-conditioned (det).
inline void ls_round(const vint* R, const vint* G, const vint* B, vint& c0, vint& c1)
{
const vint pr0 = expand5(VINT_SHIFT_RIGHT(c0, 11) & vint(31)), pg0 = expand6(VINT_SHIFT_RIGHT(c0, 5) & vint(63)), pb0 = expand5(c0 & vint(31));
const vint pr1 = expand5(VINT_SHIFT_RIGHT(c1, 11) & vint(31)), pg1 = expand6(VINT_SHIFT_RIGHT(c1, 5) & vint(63)), pb1 = expand5(c1 & vint(31));
const vint pr2 = div3(vint(2) * pr0 + pr1), pg2 = div3(vint(2) * pg0 + pg1), pb2 = div3(vint(2) * pb0 + pb1);
const vint pr3 = div3(pr0 + vint(2) * pr1), pg3 = div3(pg0 + vint(2) * pg1), pb3 = div3(pb0 + vint(2) * pb1);
const thresholds T = make_thresholds(pr0, pg0, pb0, pr1, pg1, pb1, pr2, pg2, pb2, pr3, pg3, pb3);
vfloat m00(0.0f), m01(0.0f), m11(0.0f);
vfloat a0r(0.0f), a0g(0.0f), a0b(0.0f), a1r(0.0f), a1g(0.0f), a1b(0.0f);
for (int i = 0; i < 16; i++)
{
const vint& r = R[i]; const vint& g = G[i]; const vint& b = B[i];
const vint sel = thresh_sel(T, r, g, b);
// fraction toward c1 (the LS weight b): {0, 1, 1/3, 2/3}[sel]; a = 1 - b.
const vfloat tb = spmd_ternaryf(sel == vint(0), vfloat(0.0f), spmd_ternaryf(sel == vint(1), vfloat(1.0f), spmd_ternaryf(sel == vint(2), vfloat(1.0f / 3.0f), vfloat(2.0f / 3.0f))));
const vfloat ta = vfloat(1.0f) - tb;
m00 = m00 + ta * ta; m01 = m01 + ta * tb; m11 = m11 + tb * tb;
const vfloat cr = (vfloat)r, cg = (vfloat)g, cb = (vfloat)b;
a0r = a0r + ta * cr; a0g = a0g + ta * cg; a0b = a0b + ta * cb;
a1r = a1r + tb * cr; a1g = a1g + tb * cg; a1b = a1b + tb * cb;
}
const vfloat det = m00 * m11 - m01 * m01;
const vfloat inv = vfloat(1.0f) / max(det, vfloat(1e-6f)); // clamp not bias: exact 1/det for real blocks, never /0
const vfloat e0r = (a0r * m11 - a1r * m01) * inv, e0g = (a0g * m11 - a1g * m01) * inv, e0b = (a0b * m11 - a1b * m01) * inv;
const vfloat e1r = (a1r * m00 - a0r * m01) * inv, e1g = (a1g * m00 - a0g * m01) * inv, e1b = (a1b * m00 - a0b * m01) * inv;
vint nc0 = f_to_565(e0r, e0g, e0b), nc1 = f_to_565(e1r, e1g, e1b);
{ const vint hi = max(nc0, nc1), lo = min(nc0, nc1); nc0 = hi; nc1 = lo; }
const vbool ok = det > vfloat(1e-6f); // only adopt a well-conditioned refit; degenerate lane keeps current
c0 = spmd_ternaryi(ok, nc0, c0);
c1 = spmd_ternaryi(ok, nc1, c1);
}
// Block RGB SSE for an endpoint pair (integer-threshold selectors + squared distance). For keep-better.
inline vint eval_error(const vint* R, const vint* G, const vint* B, const vint& c0, const vint& c1)
{
const vint pr0 = expand5(VINT_SHIFT_RIGHT(c0, 11) & vint(31)), pg0 = expand6(VINT_SHIFT_RIGHT(c0, 5) & vint(63)), pb0 = expand5(c0 & vint(31));
const vint pr1 = expand5(VINT_SHIFT_RIGHT(c1, 11) & vint(31)), pg1 = expand6(VINT_SHIFT_RIGHT(c1, 5) & vint(63)), pb1 = expand5(c1 & vint(31));
const vint pr2 = div3(vint(2) * pr0 + pr1), pg2 = div3(vint(2) * pg0 + pg1), pb2 = div3(vint(2) * pb0 + pb1);
const vint pr3 = div3(pr0 + vint(2) * pr1), pg3 = div3(pg0 + vint(2) * pg1), pb3 = div3(pb0 + vint(2) * pb1);
const thresholds T = make_thresholds(pr0, pg0, pb0, pr1, pg1, pb1, pr2, pg2, pb2, pr3, pg3, pb3);
vint err(0);
for (int i = 0; i < 16; i++)
{
const vint& r = R[i]; const vint& g = G[i]; const vint& b = B[i];
const vint sel = thresh_sel(T, r, g, b);
const vint psr = spmd_ternaryi(sel == vint(0), pr0, spmd_ternaryi(sel == vint(1), pr1, spmd_ternaryi(sel == vint(2), pr2, pr3)));
const vint psg = spmd_ternaryi(sel == vint(0), pg0, spmd_ternaryi(sel == vint(1), pg1, spmd_ternaryi(sel == vint(2), pg2, pg3)));
const vint psb = spmd_ternaryi(sel == vint(0), pb0, spmd_ternaryi(sel == vint(1), pb1, spmd_ternaryi(sel == vint(2), pb2, pb3)));
const vint dr = r - psr, dg = g - psg, db = b - psb;
err = err + dr * dr + dg * dg + db * db;
}
return err;
}
void _call(const basisu::color_rgba* pBlocks, uint8_t* pOut,
int* om5lo, int* om5hi, int* om6lo, int* om6hi, int ls_rounds, int avgvar,
uint32_t out_stride, uint32_t num_write)
{
const int32_t* p = (const int32_t*)pBlocks;
// Stage source pixels to SoA ONCE (lane = block): R[i]/G[i]/B[i] hold texel i of all 4 blocks.
vint R[16], G[16], B[16];
for (int i = 0; i < 16; i++) load_texel(p, i, R[i], G[i], B[i]);
// Pass 1: per-lane bounding box + channel sums + integer moments for covariance.
vint mnr(255), mng(255), mnb(255), mxr(0), mxg(0), mxb(0), sum_r(0), sum_g(0), sum_b(0);
vint sum_rr(0), sum_gg(0), sum_bb(0), sum_rg(0), sum_rb(0), sum_gb(0);
for (int i = 0; i < 16; i++)
{
const vint& r = R[i]; const vint& g = G[i]; const vint& b = B[i];
mnr = min(mnr, r); mng = min(mng, g); mnb = min(mnb, b);
mxr = max(mxr, r); mxg = max(mxg, g); mxb = max(mxb, b);
sum_r = sum_r + r; sum_g = sum_g + g; sum_b = sum_b + b;
sum_rr = sum_rr + r * r; sum_gg = sum_gg + g * g; sum_bb = sum_bb + b * b;
sum_rg = sum_rg + r * g; sum_rb = sum_rb + r * b; sum_gb = sum_gb + g * b;
}
vint color0(0), color1(0), sel_word(0);
// SOLID lanes -> omatch endpoints; all-solid groups skip the non-solid path via SPMD_SELSE's any() check.
const vbool solid = (mnr == mxr) && (mng == mxg) && (mnb == mxb);
SPMD_SIF(solid)
{
// Clone of basist::encode_bc1_solid_block: omatch endpoints from the (full-cross-product) single-color
// tables, then the swap + degenerate handling that GUARANTEES 4-color (color0 > color1), never 3-color.
const vint ar = VUINT_SHIFT_RIGHT(sum_r + vint(8), 4), ag = VUINT_SHIFT_RIGHT(sum_g + vint(8), 4), ab = VUINT_SHIFT_RIGHT(sum_b + vint(8), 4);
vint max16 = VINT_SHIFT_LEFT(load_all(ar[om5lo]), 11) | VINT_SHIFT_LEFT(load_all(ag[om6lo]), 5) | load_all(ab[om5lo]); // 2x-weighted (m_hi)
vint min16 = VINT_SHIFT_LEFT(load_all(ar[om5hi]), 11) | VINT_SHIFT_LEFT(load_all(ag[om6hi]), 5) | load_all(ab[om5hi]); // 1x (m_lo)
vint mask = vint(0xAA);
SPMD_SIF(min16 == max16) // force max16 > min16 (stay 4-color, never 3-color)
{
store(mask, vint(0));
SPMD_SIF(min16 > vint(0)) { store(min16, min16 - vint(1)); }
SPMD_SELSE(min16 > vint(0)) { store(max16, vint(1)); store(min16, vint(0)); store(mask, vint(0x55)); }
SPMD_SENDIF
}
SPMD_SENDIF
SPMD_SIF(max16 < min16) // ensure color0 > color1; selector 2<->3 flips with the swap
{
const vint a = max16, b = min16;
store(max16, b); store(min16, a);
store(mask, mask ^ vint(0x55));
}
SPMD_SENDIF
store(color0, max16); store(color1, min16);
store(sel_word, mask | VINT_SHIFT_LEFT(mask, 8) | VINT_SHIFT_LEFT(mask, 16) | VINT_SHIFT_LEFT(mask, 24));
}
SPMD_SELSE(solid)
{
// PCA seed (METHOD 2, on-axis, sqrt-free): integer-moment covariance -> no-renorm power iteration
// (bbox-diagonal seed, can't overflow at 4 iters) -> scale 1024/max -> endpoints = mean +- extent.
const vfloat cxx = (vfloat)(vint(16) * sum_rr - sum_r * sum_r), cxy = (vfloat)(vint(16) * sum_rg - sum_r * sum_g), cxz = (vfloat)(vint(16) * sum_rb - sum_r * sum_b);
const vfloat cyy = (vfloat)(vint(16) * sum_gg - sum_g * sum_g), cyz = (vfloat)(vint(16) * sum_gb - sum_g * sum_b), czz = (vfloat)(vint(16) * sum_bb - sum_b * sum_b);
vfloat ax = (vfloat)(mxr - mnr), ay = (vfloat)(mxg - mng), az = (vfloat)(mxb - mnb);
// Overflow-safe: covariance entries <= 4.16e6, 4 iters -> |v| <= ~6.2e30 << float max. Do not exceed 5 iters.
for (int it = 0; it < 4; it++)
{
const vfloat nr = ax * cxx + ay * cxy + az * cxz;
const vfloat ng = ax * cxy + ay * cyy + az * cyz;
const vfloat nb = ax * cxz + ay * cyz + az * czz;
ax = nr; ay = ng; az = nb;
}
const vfloat kk = max(max(abs(ax), abs(ay)), abs(az));
const vfloat mm = vfloat(1024.0f) / max(kk, vfloat(1e-3f)); // clamp guards the near-degenerate lane
ax = ax * mm; ay = ay * mm; az = az * mm;
const vfloat inv_len2 = vfloat(1.0f) / (ax * ax + ay * ay + az * az + vfloat(0.0000125f)); // tiny bias: never /0
const vfloat meanx = (vfloat)sum_r * vfloat(1.0f / 16.0f), meany = (vfloat)sum_g * vfloat(1.0f / 16.0f), meanz = (vfloat)sum_b * vfloat(1.0f / 16.0f);
vfloat minp(1e30f), maxp(-1e30f);
for (int i = 0; i < 16; i++)
{
const vfloat pr = ((vfloat)R[i] - meanx) * ax + ((vfloat)G[i] - meany) * ay + ((vfloat)B[i] - meanz) * az;
minp = min(minp, pr); maxp = max(maxp, pr);
}
const vfloat tlo = minp * inv_len2, thi = maxp * inv_len2;
vint c0 = f_to_565(meanx + tlo * ax, meany + tlo * ay, meanz + tlo * az);
vint c1 = f_to_565(meanx + thi * ax, meany + thi * ay, meanz + thi * az);
{ const vint hi = max(c0, c1), lo = min(c0, c1); c0 = hi; c1 = lo; }
SPMD_SIF(c0 == c1) // 4-color needs color0 > color1; nudge the degenerate lanes
{
SPMD_SIF(c1 != vint(0)) { store(c1, c1 - vint(1)); }
SPMD_SELSE(c1 != vint(0)) { store(c0, c0 + vint(1)); }
SPMD_SENDIF
}
SPMD_SENDIF
// Least-squares refinement (the big quality lever).
for (int r = 0; r < ls_rounds; r++) ls_round(R, G, B, c0, c1);
SPMD_SIF(c0 == c1)
{
SPMD_SIF(c1 != vint(0)) { store(c1, c1 - vint(1)); }
SPMD_SELSE(c1 != vint(0)) { store(c0, c0 + vint(1)); }
SPMD_SENDIF
}
SPMD_SENDIF
// Low-variance blocks (ramps) collapse the PCA seed: try an omatch-solid-of-average 2nd seed (which
// survives the collapse), 1 LS round, and keep whichever has lower error. Gated to where it's needed.
const vint var_proxy = (mxr - mnr) + (mxg - mng) + (mxb - mnb);
SPMD_SIF(var_proxy < vint(avgvar))
{
const vint ar = VUINT_SHIFT_RIGHT(sum_r + vint(8), 4), ag = VUINT_SHIFT_RIGHT(sum_g + vint(8), 4), ab = VUINT_SHIFT_RIGHT(sum_b + vint(8), 4);
vint c0b = VINT_SHIFT_LEFT(load_all(ar[om5lo]), 11) | VINT_SHIFT_LEFT(load_all(ag[om6lo]), 5) | load_all(ab[om5lo]);
vint c1b = VINT_SHIFT_LEFT(load_all(ar[om5hi]), 11) | VINT_SHIFT_LEFT(load_all(ag[om6hi]), 5) | load_all(ab[om5hi]);
{ const vint hi = max(c0b, c1b), lo = min(c0b, c1b); c0b = hi; c1b = lo; }
SPMD_SIF(c0b == c1b)
{
SPMD_SIF(c1b != vint(0)) { store(c1b, c1b - vint(1)); }
SPMD_SELSE(c1b != vint(0)) { store(c0b, c0b + vint(1)); }
SPMD_SENDIF
}
SPMD_SENDIF
ls_round(R, G, B, c0b, c1b);
SPMD_SIF(c0b == c1b)
{
SPMD_SIF(c1b != vint(0)) { store(c1b, c1b - vint(1)); }
SPMD_SELSE(c1b != vint(0)) { store(c0b, c0b + vint(1)); }
SPMD_SENDIF
}
SPMD_SENDIF
const vbool use_avg = eval_error(R, G, B, c0b, c1b) < eval_error(R, G, B, c0, c1);
store(c0, spmd_ternaryi(use_avg, c0b, c0));
store(c1, spmd_ternaryi(use_avg, c1b, c1));
}
SPMD_SENDIF
// Final palette + integer-threshold selector pass.
const vint pr0 = expand5(VINT_SHIFT_RIGHT(c0, 11) & vint(31)), pg0 = expand6(VINT_SHIFT_RIGHT(c0, 5) & vint(63)), pb0 = expand5(c0 & vint(31));
const vint pr1 = expand5(VINT_SHIFT_RIGHT(c1, 11) & vint(31)), pg1 = expand6(VINT_SHIFT_RIGHT(c1, 5) & vint(63)), pb1 = expand5(c1 & vint(31));
const vint pr2 = div3(vint(2) * pr0 + pr1), pg2 = div3(vint(2) * pg0 + pg1), pb2 = div3(vint(2) * pb0 + pb1);
const vint pr3 = div3(pr0 + vint(2) * pr1), pg3 = div3(pg0 + vint(2) * pg1), pb3 = div3(pb0 + vint(2) * pb1);
const thresholds T = make_thresholds(pr0, pg0, pb0, pr1, pg1, pb1, pr2, pg2, pb2, pr3, pg3, pb3);
vint sw(0);
for (int i = 0; i < 16; i++)
{
const vint bestsel = thresh_sel(T, R[i], G[i], B[i]);
sw = sw | (bestsel << (i * 2));
}
store(color0, c0); store(color1, c1); store(sel_word, sw);
}
SPMD_SENDIF
// Emit one BC1 block per lane directly to the (possibly strided) output -- no temp buffer. num_write < 4
// for the final partial group so the padded lanes aren't written; out_stride lets BC3/BC5 interleave.
CPPSPMD_DECL(int, c0a[4]); CPPSPMD_DECL(int, c1a[4]); CPPSPMD_DECL(int, swa[4]);
storeu_linear_all(c0a, color0);
storeu_linear_all(c1a, color1);
storeu_linear_all(swa, sel_word);
for (uint32_t j = 0; j < num_write; j++)
{
uint8_t* o = pOut + j * out_stride;
const uint32_t c0 = (uint32_t)c0a[j], c1 = (uint32_t)c1a[j], sw = (uint32_t)swa[j];
o[0] = (uint8_t)c0; o[1] = (uint8_t)(c0 >> 8);
o[2] = (uint8_t)c1; o[3] = (uint8_t)(c1 >> 8);
o[4] = (uint8_t)sw; o[5] = (uint8_t)(sw >> 8); o[6] = (uint8_t)(sw >> 16); o[7] = (uint8_t)(sw >> 24);
}
}
};
} // namespace bc_spmd_kern
namespace bc_spmd_kern
{
// BC4 (RGTC1) -- one single-channel block per lane, 4 blocks/call. Clone of basist::encode_bc4: raw bbox
// min/max, exact nearest-of-8 integer-threshold selector, ALWAYS 8-value mode (solid forced to red0>red1).
// All-integer, GATHER-FREE: texels via manual load; rank->index by arithmetic (no s_tran gather); the 48-bit
// selector packed via uniform shifts into two vints (texels 0-9 in lo, 10-15 in hi), combined with <<30 in scalar.
struct encode_bc4_blocks : spmd_kernel
{
// Nearest-of-8 selectors + decoded SSE for an 8-value endpoint pair (r0 > r1). Packs texels 0-9 into slo,
// 10-15 into shi. Optionally accumulates the 1-D least-squares moments (rank = interpolation weight 0..7).
inline vint bc4_eval(const vint* A, const vint& r0, const vint& r1, vint& slo, vint& shi, vint* pSw, vint* pSww, vint* pSv, vint* pSwv)
{
const vint delta = r0 - r1;
const vint bias = vint(4) - r1 * vint(14);
const vint t0 = delta * vint(13), t1 = delta * vint(11), t2 = delta * vint(9), t3 = delta * vint(7), t4 = delta * vint(5), t5 = delta * vint(3), t6 = delta;
vint sse(0), Sw(0), Sww(0), Sv(0), Swv(0);
slo = vint(0); shi = vint(0);
for (int i = 0; i < 16; i++)
{
const vint val = A[i];
const vint sv = val * vint(14) + bias;
const vint rank = spmd_ternaryi(sv >= t0, 1, 0) + spmd_ternaryi(sv >= t1, 1, 0) + spmd_ternaryi(sv >= t2, 1, 0)
+ spmd_ternaryi(sv >= t3, 1, 0) + spmd_ternaryi(sv >= t4, 1, 0) + spmd_ternaryi(sv >= t5, 1, 0) + spmd_ternaryi(sv >= t6, 1, 0);
const vint idx = spmd_ternaryi(rank == vint(7), vint(0), spmd_ternaryi(rank == vint(0), vint(1), vint(8) - rank));
if (i < 10) slo = slo | (idx << (i * 3));
else shi = shi | (idx << ((i - 10) * 3));
const vint num = rank * r0 + (vint(7) - rank) * r1;
const vint recon = VUINT_SHIFT_RIGHT(num * vint(9363), 16); // == num/7 floor for num<=1785 (matches scalar /7)
const vint d = val - recon; sse = sse + d * d;
Sw = Sw + rank; Sww = Sww + rank * rank; Sv = Sv + val; Swv = Swv + rank * val;
}
if (pSw) { *pSw = Sw; *pSww = Sww; *pSv = Sv; *pSwv = Swv; }
return sse;
}
void _call(const uint8_t* pBlocks, uint8_t* pOut, uint32_t out_stride, uint32_t num_write, uint32_t do_ls, uint32_t src_stride)
{
// Stage 16 channel values to SoA: A[i] = texel i of the 4 lane-blocks. Source values are src_stride bytes
// apart (1 = packed channel, 4 = one channel of RGBA); blocks are 16*src_stride bytes apart.
const uint32_t S = src_stride;
vint A[16];
for (int i = 0; i < 16; i++)
A[i].m_value = _mm_setr_epi32(pBlocks[i * S], pBlocks[(16 + i) * S], pBlocks[(32 + i) * S], pBlocks[(48 + i) * S]);
vint mn = A[0], mx = A[0];
for (int i = 1; i < 16; i++) { mn = min(mn, A[i]); mx = max(mx, A[i]); }
vint color0(0), color1(0), sel_lo(0), sel_hi(0);
const vbool solid = (mx == mn);
SPMD_SIF(solid)
{
// Force 8-value solid: value>0 -> (max, max-1, all idx 0); value==0 -> (1, 0, all idx 1).
const vbool pos = (mx > vint(0));
store(color0, spmd_ternaryi(pos, mx, vint(1)));
store(color1, spmd_ternaryi(pos, mx - vint(1), vint(0)));
const vint idx = spmd_ternaryi(pos, vint(0), vint(1)); // replicated into every 3-bit field
store(sel_lo, idx * vint(0x9249249)); // 10 fields (texels 0-9) all = idx
store(sel_hi, idx * vint(0x9249)); // 6 fields (texels 10-15) all = idx
}
SPMD_SELSE(solid)
{
if (do_ls)
{
// HQ: candidate A = raw bbox (== encode_bc4) + LS moments; candidate B = 1-D LS endpoint refit
// (value ~= a + (rank/7)*b). detf guards /0 for degenerate lanes (refit then rejected). Keep-best.
vint sloA, shiA, Sw, Sww, Sv, Swv;
const vint sseA = bc4_eval(A, mx, mn, sloA, shiA, &Sw, &Sww, &Sv, &Swv);
const vint detI = vint(16) * Sww - Sw * Sw;
const vfloat detf = (vfloat)spmd_ternaryi(detI == vint(0), vint(1), detI);
const vfloat b = vfloat(7.0f) * (vfloat)(vint(16) * Swv - Sw * Sv) / detf; // slope = red0 - red1
const vfloat bsw7 = b * (vfloat)Sw * vfloat(1.0f / 7.0f);
const vfloat a = ((vfloat)Sv - bsw7) / vfloat(16.0f); // intercept = red1
vint nr1 = vint(round_nearest(a)); nr1 = min(max(nr1, vint(0)), vint(255));
vint nr0 = vint(round_nearest(a + b)); nr0 = min(max(nr0, vint(0)), vint(255));
vint sloB, shiB;
const vint sseB = bc4_eval(A, nr0, nr1, sloB, shiB, nullptr, nullptr, nullptr, nullptr);
const vbool adopt = ((detI != vint(0)) && (nr0 > nr1)) && (sseB < sseA);
store(color0, spmd_ternaryi(adopt, nr0, mx));
store(color1, spmd_ternaryi(adopt, nr1, mn));
store(sel_lo, spmd_ternaryi(adopt, sloB, sloA));
store(sel_hi, spmd_ternaryi(adopt, shiB, shiA));
}
else
{
// FAST: raw bbox endpoints + nearest-of-8 selector only (== basist::encode_bc4 quality, ~1.4x).
store(color0, mx); store(color1, mn);
const vint delta = mx - mn;
const vint bias = vint(4) - mn * vint(14);
const vint t0 = delta * vint(13), t1 = delta * vint(11), t2 = delta * vint(9), t3 = delta * vint(7), t4 = delta * vint(5), t5 = delta * vint(3), t6 = delta;
vint slo(0), shi(0);
for (int i = 0; i < 16; i++)
{
const vint sv = A[i] * vint(14) + bias;
const vint rank = spmd_ternaryi(sv >= t0, 1, 0) + spmd_ternaryi(sv >= t1, 1, 0) + spmd_ternaryi(sv >= t2, 1, 0)
+ spmd_ternaryi(sv >= t3, 1, 0) + spmd_ternaryi(sv >= t4, 1, 0) + spmd_ternaryi(sv >= t5, 1, 0) + spmd_ternaryi(sv >= t6, 1, 0);
const vint idx = spmd_ternaryi(rank == vint(7), vint(0), spmd_ternaryi(rank == vint(0), vint(1), vint(8) - rank));
if (i < 10) slo = slo | (idx << (i * 3));
else shi = shi | (idx << ((i - 10) * 3));
}
store(sel_lo, slo); store(sel_hi, shi);
}
}
SPMD_SENDIF
CPPSPMD_DECL(int, c0a[4]); CPPSPMD_DECL(int, c1a[4]); CPPSPMD_DECL(int, sloa[4]); CPPSPMD_DECL(int, shia[4]);
storeu_linear_all(c0a, color0);
storeu_linear_all(c1a, color1);
storeu_linear_all(sloa, sel_lo);
storeu_linear_all(shia, sel_hi);
for (uint32_t j = 0; j < num_write; j++)
{
uint8_t* o = pOut + j * out_stride;
o[0] = (uint8_t)c0a[j]; o[1] = (uint8_t)c1a[j];
const uint64_t W = (uint64_t)(uint32_t)sloa[j] | ((uint64_t)(uint32_t)shia[j] << 30); // 48-bit selector word
o[2] = (uint8_t)W; o[3] = (uint8_t)(W >> 8); o[4] = (uint8_t)(W >> 16);
o[5] = (uint8_t)(W >> 24); o[6] = (uint8_t)(W >> 32); o[7] = (uint8_t)(W >> 40);
}
}
};
} // namespace bc_spmd_kern
namespace basisu
{
namespace bc_spmd
{
void encode_bc4_blocks_4_sse41(const uint8_t* pBlocks, uint8_t* pOut, uint32_t out_stride, uint32_t num_write, uint32_t do_ls, uint32_t src_stride)
{
spmd_call<bc_spmd_kern::encode_bc4_blocks>(pBlocks, pOut, out_stride, num_write, do_ls, src_stride);
}
}
}
// ISA wrapper -- encode 4 blocks at once. Defined here (inside the SSE4.1 TU), declared in basisu_bc15_spmd.cpp.
namespace basisu
{
namespace bc_spmd
{
void encode_bc1_blocks_4_sse41(const color_rgba* pBlocks, uint8_t* pOut,
int* om5lo, int* om5hi, int* om6lo, int* om6hi, int ls_rounds, int avgvar,
uint32_t out_stride, uint32_t num_write)
{
spmd_call<bc_spmd_kern::encode_bc1_4color_blocks>(pBlocks, pOut, om5lo, om5hi, om6lo, om6hi, ls_rounds, avgvar, out_stride, num_write);
}
}
}

View File

@@ -0,0 +1,22 @@
// basisu_bc15_spmd_sse.cpp -- SSE4.1 ISA translation unit for the standalone BC1 encoder.
//
// Mirrors basisu_kernels_sse.cpp's pattern: select the ISA, include the cppspmd framework, then pull in the
// kernel .inl ("do not directly include"). Multiple TUs may include the framework because cppspmd_math.h's
// print_* helpers are inline.
#include "basisu_enc.h" // basisu::color_rgba + BASISU_SUPPORT_SSE
#if BASISU_SUPPORT_SSE
#define CPPSPMD_SSE2 0
#ifdef _MSC_VER
#include <intrin.h>
#endif
#include "cppspmd_sse.h"
#include "cppspmd_type_aliases.h"
using namespace CPPSPMD;
#include "basisu_bc15_spmd_kernels.inl"
#endif // BASISU_SUPPORT_SSE

File diff suppressed because it is too large Load Diff

View File

@@ -0,0 +1,94 @@
// bc7e_scalar.h - Pure scalar C++17 port of bc7e.ispc (no ISPC, no SIMD).
// Drop-in alternative BC7 encoder exposing the same surface as the ISPC build,
// but in the bc7e_scalar:: namespace so it can run side-by-side with ispc:: for
// quality comparison. See bc7e.ispc for the original; this is de-SIMD'd to
// encode one 4x4 block at a time.
#pragma once
#include <stdint.h>
namespace bc7e_scalar
{
// Mirrors ispc::bc7e_compress_block_params (identical layout/semantics).
struct bc7e_compress_block_params
{
uint32_t m_max_partitions_mode[8];
uint32_t m_weights[4];
uint32_t m_uber_level;
uint32_t m_refinement_passes;
uint32_t m_mode4_rotation_mask;
uint32_t m_mode4_index_mask;
uint32_t m_mode5_rotation_mask;
uint32_t m_uber1_mask;
bool m_perceptual;
bool m_pbit_search;
bool m_mode6_only;
// When false, color_cell_compression() is NOT allowed to use the precomputed
// "one color" optimal-endpoint lookup tables (the solid/allSame and average-
// color paths). Disabling them avoids the extreme-endpoint "trap" blocks that
// decode badly under lossy weight coding. Default TRUE (matches bc7e's original
// behavior); callers disable it when needed (e.g. XBC7 below lossless Q).
// (Repurposed from the former m_unused0 padding bool.)
bool m_use_luts;
struct
{
uint32_t m_max_mode13_partitions_to_try;
uint32_t m_max_mode0_partitions_to_try;
uint32_t m_max_mode2_partitions_to_try;
bool m_use_mode[7];
bool m_unused1;
} m_opaque_settings;
struct
{
uint32_t m_max_mode7_partitions_to_try;
uint32_t m_mode67_error_weight_mul[4];
bool m_use_mode4;
bool m_use_mode5;
bool m_use_mode6;
bool m_use_mode7;
bool m_use_mode4_rotation;
bool m_use_mode5_rotation;
bool m_unused2;
bool m_unused3;
} m_alpha_settings;
};
void bc7e_compress_block_init();
void bc7e_compress_block_params_init(bc7e_compress_block_params* p, bool perceptual);
void bc7e_compress_block_params_init_basic(bc7e_compress_block_params* p, bool perceptual);
void bc7e_compress_block_params_init_fast(bc7e_compress_block_params* p, bool perceptual);
void bc7e_compress_block_params_init_slow(bc7e_compress_block_params* p, bool perceptual);
void bc7e_compress_block_params_init_slowest(bc7e_compress_block_params* p, bool perceptual);
void bc7e_compress_block_params_init_ultrafast(bc7e_compress_block_params* p, bool perceptual);
void bc7e_compress_block_params_init_veryfast(bc7e_compress_block_params* p, bool perceptual);
void bc7e_compress_block_params_init_veryslow(bc7e_compress_block_params* p, bool perceptual);
// pUsed_lut (optional, one byte per block): set to 1 if the WINNING encoding of that
// block used the precomputed "one color" optimal-endpoint lookup tables on any subset
// (solid/allSame or average-color path) -- a hint that the block placed endpoints at
// extreme positions and may be "weird"/fragile under lossy weight recoding. 0 otherwise.
void bc7e_compress_blocks(uint32_t num_blocks, uint64_t* pBlocks, const uint32_t* pPixelsRGBA, const bc7e_compress_block_params* pComp_params, uint8_t* pUsed_lut = nullptr);
// Compress a single 4x4 RGBA block (16 pixels) to ONE specified BC7 mode only, using
// the given settings. Writes the 16-byte block to pBlock and returns its encoding error.
// Additive testing API - does not affect bc7e_compress_blocks().
// mode : 0..7
// partition : modes 0,1,2,3,7 -> partition pattern index; -1 = auto-select optimal
// (existing logic). Ignored for modes 4,5,6.
// rotation : modes 4,5 -> dual-plane component rotation [0..3] (0 = none). A non-zero
// rotation forces linear (non-perceptual) error internally. Ignored otherwise.
// index_selector : mode 4 -> 0 or 1 index-set selector. Ignored otherwise.
// Note: forcing an opaque mode (0-3) on a block with alpha drops alpha (A becomes 255).
uint64_t bc7e_compress_block_single_mode(uint64_t* pBlock, const uint32_t* pPixelsRGBA, const bc7e_compress_block_params* pComp_params,
uint32_t mode, int partition = -1, uint32_t rotation = 0, uint32_t index_selector = 0);
} // namespace bc7e_scalar

View File

@@ -0,0 +1,569 @@
// basisu_dds_export.cpp
// Copyright (C) 2019-2025 Binomial LLC. All Rights Reserved.
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//
// See basisu_dds_export.h for an overview.
#include "basisu_dds_export.h"
#include "basisu_comp.h"
#include "basisu_bc7e_scalar.h"
#include "../transcoder/basisu_transcoder_internal.h"
#include "../transcoder/basisu_transcoder_uastc.h"
namespace basisu
{
// --- DXGI format codes (only the ones we emit). ---
enum
{
DXGI_FORMAT_R8G8B8A8_UNORM = 28,
DXGI_FORMAT_R8G8B8A8_UNORM_SRGB = 29,
DXGI_FORMAT_R8G8_UNORM = 49,
DXGI_FORMAT_R8_UNORM = 61,
DXGI_FORMAT_BC1_UNORM = 71,
DXGI_FORMAT_BC1_UNORM_SRGB = 72,
DXGI_FORMAT_BC2_UNORM = 74,
DXGI_FORMAT_BC2_UNORM_SRGB = 75,
DXGI_FORMAT_BC3_UNORM = 77,
DXGI_FORMAT_BC3_UNORM_SRGB = 78,
DXGI_FORMAT_BC4_UNORM = 80,
DXGI_FORMAT_BC5_UNORM = 83,
DXGI_FORMAT_B5G6R5_UNORM = 85,
DXGI_FORMAT_B5G5R5A1_UNORM = 86,
DXGI_FORMAT_B8G8R8A8_UNORM = 87,
DXGI_FORMAT_B8G8R8A8_UNORM_SRGB = 91,
DXGI_FORMAT_BC7_UNORM = 98,
DXGI_FORMAT_BC7_UNORM_SRGB = 99,
DXGI_FORMAT_B4G4R4A4_UNORM = 115
};
// --- DDS header flag/cap constants. ---
enum
{
DDSD_CAPS = 0x1, DDSD_HEIGHT = 0x2, DDSD_WIDTH = 0x4, DDSD_PITCH = 0x8,
DDSD_PIXELFORMAT = 0x1000, DDSD_MIPMAPCOUNT = 0x20000, DDSD_LINEARSIZE = 0x80000,
DDPF_FOURCC = 0x4,
DDSCAPS_COMPLEX = 0x8, DDSCAPS_TEXTURE = 0x1000, DDSCAPS_MIPMAP = 0x400000,
DDSCAPS2_CUBEMAP = 0x200, DDSCAPS2_CUBEMAP_ALLFACES = 0xFC00,
DDS_MAGIC = 0x20534444, // "DDS "
DDS_DX10_FOURCC = 0x30315844, // "DX10"
DDS_DIMENSION_TEXTURE2D = 3,
DDS_RESOURCE_MISC_TEXTURECUBE = 0x4
};
// Per-format static info.
struct dds_format_info
{
const char* m_pToken;
bool m_block_compressed;
uint32_t m_bytes; // bytes per 4x4 block (compressed) or bytes per pixel (uncompressed)
uint32_t m_dxgi_unorm;
uint32_t m_dxgi_srgb; // 0 if no sRGB variant exists
};
static const dds_format_info g_dds_format_info[cDDSFmtTotal] =
{
// token block bytes unorm srgb
{ "bc1", true, 8, DXGI_FORMAT_BC1_UNORM, DXGI_FORMAT_BC1_UNORM_SRGB },
{ "bc2", true, 16, DXGI_FORMAT_BC2_UNORM, DXGI_FORMAT_BC2_UNORM_SRGB },
{ "bc3", true, 16, DXGI_FORMAT_BC3_UNORM, DXGI_FORMAT_BC3_UNORM_SRGB },
{ "bc4", true, 8, DXGI_FORMAT_BC4_UNORM, 0 },
{ "bc5", true, 16, DXGI_FORMAT_BC5_UNORM, 0 },
{ "bc7", true, 16, DXGI_FORMAT_BC7_UNORM, DXGI_FORMAT_BC7_UNORM_SRGB },
{ "a8r8g8b8", false, 4, DXGI_FORMAT_B8G8R8A8_UNORM, DXGI_FORMAT_B8G8R8A8_UNORM_SRGB },
{ "a8b8g8r8", false, 4, DXGI_FORMAT_R8G8B8A8_UNORM, DXGI_FORMAT_R8G8B8A8_UNORM_SRGB },
{ "r8", false, 1, DXGI_FORMAT_R8_UNORM, 0 },
{ "r8g8", false, 2, DXGI_FORMAT_R8G8_UNORM, 0 },
{ "r5g6b5", false, 2, DXGI_FORMAT_B5G6R5_UNORM, 0 },
{ "a1r5g5b5", false, 2, DXGI_FORMAT_B5G5R5A1_UNORM, 0 },
{ "a4r4g4b4", false, 2, DXGI_FORMAT_B4G4R4A4_UNORM, 0 }
};
bool parse_dds_output_format(const char* pToken, dds_output_format& fmt)
{
fmt = cDDSFmtInvalid;
if (!pToken)
return false;
for (int i = 0; i < (int)cDDSFmtTotal; i++)
{
if (strcasecmp(pToken, g_dds_format_info[i].m_pToken) == 0)
{
fmt = (dds_output_format)i;
return true;
}
}
// Tolerate the (geometrically impossible) "a1r5g6b5" spelling as an alias for a1r5g5b5.
if (strcasecmp(pToken, "a1r5g6b5") == 0)
{
fmt = cDDSFmtA1R5G5B5;
return true;
}
return false;
}
const char* get_dds_output_format_string(dds_output_format fmt)
{
if ((fmt < 0) || (fmt >= cDDSFmtTotal))
return "?";
return g_dds_format_info[fmt].m_pToken;
}
bool dds_output_format_has_srgb_variant(dds_output_format fmt)
{
if ((fmt < 0) || (fmt >= cDDSFmtTotal))
return false;
return g_dds_format_info[fmt].m_dxgi_srgb != 0;
}
// --- Little-endian append helpers. ---
static inline void append_u16(uint8_vec& v, uint32_t x)
{
v.push_back((uint8_t)(x & 0xFF));
v.push_back((uint8_t)((x >> 8) & 0xFF));
}
static inline void append_u32(uint8_vec& v, uint32_t x)
{
v.push_back((uint8_t)(x & 0xFF));
v.push_back((uint8_t)((x >> 8) & 0xFF));
v.push_back((uint8_t)((x >> 16) & 0xFF));
v.push_back((uint8_t)((x >> 24) & 0xFF));
}
// Packs a single RGBA8 texel into the uncompressed output bytes for fmt (truncating 8->N bits, which is
// the exact inverse of the reader's bit-replication expansion). Channel byte orders match the DXGI formats.
static void pack_uncompressed_pixel(uint8_vec& out, const color_rgba& c, dds_output_format fmt)
{
switch (fmt)
{
case cDDSFmtA8R8G8B8: // DXGI B8G8R8A8 -> memory order B,G,R,A
out.push_back(c.b); out.push_back(c.g); out.push_back(c.r); out.push_back(c.a);
break;
case cDDSFmtA8B8G8R8: // DXGI R8G8B8A8 -> memory order R,G,B,A
out.push_back(c.r); out.push_back(c.g); out.push_back(c.b); out.push_back(c.a);
break;
case cDDSFmtR8: // DXGI R8 -> just R
out.push_back(c.r);
break;
case cDDSFmtR8G8: // DXGI R8G8 -> source R into R, source G into G (matches BC5; swizzle input for other mappings)
out.push_back(c.r); out.push_back(c.g);
break;
case cDDSFmtR5G6B5:
append_u16(out, (uint32_t)(((c.r >> 3) << 11) | ((c.g >> 2) << 5) | (c.b >> 3)));
break;
case cDDSFmtA1R5G5B5:
append_u16(out, (uint32_t)((c.a >= 128 ? 0x8000 : 0) | ((c.r >> 3) << 10) | ((c.g >> 3) << 5) | (c.b >> 3)));
break;
case cDDSFmtA4R4G4B4:
append_u16(out, (uint32_t)(((c.a >> 4) << 12) | ((c.r >> 4) << 8) | ((c.g >> 4) << 4) | (c.b >> 4)));
break;
default:
assert(0);
break;
}
}
// Prebuilt BC7 packing context (built once per build_dds, shared read-only across slices).
struct bc7_pack_context
{
dds_bc7_encoder m_encoder;
uint32_t m_bc7f_flags; // bc7f packer flags (when m_encoder == bc7f)
bc7e_scalar::bc7e_compress_block_params m_bc7e_params; // initialized only when m_encoder == bc7e_scalar
};
// Packs one prepared slice image into the bytes for fmt, using logical dims (orig_width/orig_height).
// For block formats this iterates whole 4x4 blocks (the slice image is already block-padded); for
// uncompressed it emits tight rows of exactly orig_width*orig_height pixels.
static void pack_slice(uint8_vec& out, const image& img, uint32_t orig_width, uint32_t orig_height, dds_output_format fmt, const bc7_pack_context& bc7ctx)
{
const dds_format_info& info = g_dds_format_info[fmt];
if (info.m_block_compressed)
{
const uint32_t blocks_x = (orig_width + 3) / 4;
const uint32_t blocks_y = (orig_height + 3) / 4;
for (uint32_t by = 0; by < blocks_y; by++)
{
for (uint32_t bx = 0; bx < blocks_x; bx++)
{
color_rgba blk[16];
img.extract_block_clamped(blk, bx * 4, by * 4, 4, 4);
uint8_t dst[16];
switch (fmt)
{
case cDDSFmtBC1:
basist::encode_bc1(dst, (const uint8_t*)blk, basist::cEncodeBC1HighQuality);
break;
case cDDSFmtBC2:
// 8 bytes explicit 4-bit alpha (one LE 16-bit word per row, texel x at nibble x), then a
// BC1 color block (encode_bc1 emits 4-color blocks, which BC2's always-4-color decode needs).
for (uint32_t ry = 0; ry < 4; ry++)
{
uint32_t row = 0;
for (uint32_t rx = 0; rx < 4; rx++)
row |= ((uint32_t)(blk[ry * 4 + rx].a >> 4)) << (rx * 4);
dst[ry * 2] = (uint8_t)(row & 0xFF);
dst[ry * 2 + 1] = (uint8_t)(row >> 8);
}
basist::encode_bc1(dst + 8, (const uint8_t*)blk, basist::cEncodeBC1HighQuality);
break;
case cDDSFmtBC3:
basist::encode_bc4(dst, &blk[0].a, sizeof(color_rgba));
basist::encode_bc1(dst + 8, (const uint8_t*)blk, basist::cEncodeBC1HighQuality);
break;
case cDDSFmtBC4:
basist::encode_bc4(dst, &blk[0].r, sizeof(color_rgba));
break;
case cDDSFmtBC5:
basist::encode_bc4(dst, &blk[0].r, sizeof(color_rgba));
basist::encode_bc4(dst + 8, &blk[0].g, sizeof(color_rgba));
break;
case cDDSFmtBC7:
// basist::color_rgba and basisu::color_rgba have identical layout (r,g,b,a uint8_t).
if (bc7ctx.m_encoder == cDDSBC7Encoder_BC7E_Scalar)
{
uint64_t blk64[2];
bc7e_scalar::bc7e_compress_blocks(1, blk64, reinterpret_cast<const uint32_t*>(blk), &bc7ctx.m_bc7e_params, nullptr);
memcpy(dst, blk64, 16);
}
else
{
basist::bc7f::fast_pack_bc7_auto_rgba(dst, reinterpret_cast<const basist::color_rgba*>(blk), bc7ctx.m_bc7f_flags);
}
break;
default:
assert(0);
break;
}
out.append(dst, info.m_bytes);
}
}
}
else
{
for (uint32_t y = 0; y < orig_height; y++)
for (uint32_t x = 0; x < orig_width; x++)
pack_uncompressed_pixel(out, img(x, y), fmt);
}
}
// Builds the DDS magic + DDS_HEADER + DDS_HEADER_DXT10 (148 bytes total).
static void build_dx10_header(uint8_vec& out, uint32_t width, uint32_t height, uint32_t mip_count,
uint32_t array_size, bool is_cubemap, uint32_t dxgi_format, const dds_format_info& info)
{
// dwPitchOrLinearSize (informational; readers generally recompute).
uint32_t pitch_or_linear_size;
uint32_t flags = DDSD_CAPS | DDSD_HEIGHT | DDSD_WIDTH | DDSD_PIXELFORMAT;
if (info.m_block_compressed)
{
pitch_or_linear_size = ((width + 3) / 4) * ((height + 3) / 4) * info.m_bytes;
flags |= DDSD_LINEARSIZE;
}
else
{
pitch_or_linear_size = width * info.m_bytes;
flags |= DDSD_PITCH;
}
if (mip_count > 1)
flags |= DDSD_MIPMAPCOUNT;
uint32_t caps = DDSCAPS_TEXTURE;
if ((mip_count > 1) || is_cubemap || (array_size > 1))
caps |= DDSCAPS_COMPLEX;
if (mip_count > 1)
caps |= DDSCAPS_MIPMAP;
const uint32_t caps2 = is_cubemap ? (DDSCAPS2_CUBEMAP | DDSCAPS2_CUBEMAP_ALLFACES) : 0;
append_u32(out, DDS_MAGIC);
// DDS_HEADER (124 bytes).
append_u32(out, 124); // dwSize
append_u32(out, flags); // dwFlags
append_u32(out, height); // dwHeight
append_u32(out, width); // dwWidth
append_u32(out, pitch_or_linear_size);
append_u32(out, 0); // dwDepth
append_u32(out, mip_count); // dwMipMapCount
for (uint32_t i = 0; i < 11; i++) // dwReserved1[11]
append_u32(out, 0);
// DDS_PIXELFORMAT (32 bytes) - DX10 indirection.
append_u32(out, 32); // ddspf.dwSize
append_u32(out, DDPF_FOURCC); // ddspf.dwFlags
append_u32(out, DDS_DX10_FOURCC); // ddspf.dwFourCC = "DX10"
append_u32(out, 0); // dwRGBBitCount
append_u32(out, 0); // dwRBitMask
append_u32(out, 0); // dwGBitMask
append_u32(out, 0); // dwBBitMask
append_u32(out, 0); // dwABitMask
append_u32(out, caps); // dwCaps
append_u32(out, caps2); // dwCaps2
append_u32(out, 0); // dwCaps3
append_u32(out, 0); // dwCaps4
append_u32(out, 0); // dwReserved2
// DDS_HEADER_DXT10 (20 bytes).
append_u32(out, dxgi_format); // dxgiFormat
append_u32(out, DDS_DIMENSION_TEXTURE2D);
append_u32(out, is_cubemap ? DDS_RESOURCE_MISC_TEXTURECUBE : 0); // miscFlag
append_u32(out, array_size); // arraySize (number of array elements; cubes for a cubemap)
append_u32(out, 0); // miscFlags2
}
bool build_dds(uint8_vec& dds_data, const basis_compressor& comp, const dds_export_params& params,
std::string& error_msg,
uint32_t* pOut_width, uint32_t* pOut_height,
uint32_t* pOut_levels, uint32_t* pOut_layers, uint32_t* pOut_faces)
{
error_msg.clear();
dds_data.resize(0);
const dds_output_format fmt = params.m_format;
if ((fmt < 0) || (fmt >= cDDSFmtTotal))
{
error_msg = "invalid output format";
return false;
}
const dds_format_info& info = g_dds_format_info[fmt];
const basisu::vector<image>& slices = comp.get_slice_images();
const basisu_backend_slice_desc_vec& descs = comp.get_slice_descs();
if (!slices.size() || (slices.size() != descs.size()))
{
error_msg = "no prepared slices (was process_source_images() run successfully?)";
return false;
}
const bool is_cubemap = (comp.get_params().m_tex_type == basist::cBASISTexTypeCubemapArray);
// Determine base dims, layer count, mip level count, and face count - mirrors create_ktx2_file().
uint32_t base_width = 0, base_height = 0, total_layers = 0, total_levels = 0, total_faces = 1;
for (uint32_t i = 0; i < descs.size(); i++)
{
if ((descs[i].m_mip_index == 0) && (!base_width))
{
base_width = descs[i].m_orig_width;
base_height = descs[i].m_orig_height;
}
total_layers = maximum<uint32_t>(total_layers, descs[i].m_source_file_index + 1);
if (!descs[i].m_source_file_index)
total_levels = maximum<uint32_t>(total_levels, descs[i].m_mip_index + 1);
}
if (is_cubemap)
{
if ((total_layers % 6) != 0)
{
error_msg = "cubemap source image count is not a multiple of 6";
return false;
}
total_layers /= 6;
total_faces = 6;
}
if (!base_width || !base_height || !total_layers || !total_levels)
{
error_msg = "could not determine texture dimensions from the prepared slices";
return false;
}
// Build a (layer, face, level) -> slice index map using the same decomposition as create_ktx2_file().
const uint32_t total_subresources = total_layers * total_faces * total_levels;
basisu::vector<int> slice_map(total_subresources);
for (uint32_t i = 0; i < total_subresources; i++)
slice_map[i] = -1;
for (uint32_t i = 0; i < descs.size(); i++)
{
// Note: descs[i].m_alpha just flags that the slice contains alpha; in the XUBC7 (m_uastc) path each
// slice holds full RGBA, so it is NOT a separate alpha-only slice (that only happens for ETC1S).
const uint32_t level_index = descs[i].m_mip_index;
uint32_t layer_index = descs[i].m_source_file_index;
uint32_t face_index = 0;
if (is_cubemap)
{
face_index = layer_index % 6;
layer_index /= 6;
}
if ((layer_index >= total_layers) || (face_index >= total_faces) || (level_index >= total_levels))
{
error_msg = "slice descriptor out of range (internal error)";
return false;
}
const uint32_t map_index = (layer_index * total_faces + face_index) * total_levels + level_index;
// Two slices mapping to the same subresource would indicate RGB/alpha slice splitting (ETC1S), which
// we don't expect here since we force the XUBC7 path.
if (slice_map[map_index] >= 0)
{
error_msg = "multiple slices map to the same (layer, face, level) - RGB/alpha slice splitting is not supported";
return false;
}
slice_map[map_index] = (int)i;
}
for (uint32_t i = 0; i < total_subresources; i++)
{
if (slice_map[i] < 0)
{
error_msg = "missing slice for a (layer, face, level) - source images are not all the same size / mip count";
return false;
}
}
// Select the DXGI variant (UNORM vs sRGB).
const bool want_srgb = comp.get_params().m_ktx2_and_basis_srgb_transfer_function;
const uint32_t dxgi_format = (want_srgb && info.m_dxgi_srgb) ? info.m_dxgi_srgb : info.m_dxgi_unorm;
if (want_srgb && !info.m_dxgi_srgb && comp.get_params().m_status_output)
printf("Note: format %s has no sRGB DXGI variant; writing UNORM (texel data is unchanged regardless).\n", info.m_pToken);
// Build the BC7 packing context once (only relevant for the bc7 format).
bc7_pack_context bc7ctx;
memset(&bc7ctx, 0, sizeof(bc7ctx));
bc7ctx.m_encoder = params.m_bc7_encoder;
switch (clamp<int>(params.m_bc7f_level, cDDSBC7FLevel_Analytical, cDDSBC7FLevel_NonAnalytical))
{
case cDDSBC7FLevel_Analytical: bc7ctx.m_bc7f_flags = basist::bc7f::cPackBC7FlagDefault; break;
case cDDSBC7FLevel_NonAnalytical: bc7ctx.m_bc7f_flags = basist::bc7f::cPackBC7FlagDefaultNonAnalytical; break;
default: bc7ctx.m_bc7f_flags = basist::bc7f::cPackBC7FlagDefaultPartiallyAnalytical; break;
}
if ((fmt == cDDSFmtBC7) && (params.m_bc7_encoder == cDDSBC7Encoder_BC7E_Scalar))
{
bc7e_scalar::bc7e_compress_block_init();
typedef void (*bc7e_init_func)(bc7e_scalar::bc7e_compress_block_params*, bool);
static const bc7e_init_func s_bc7e_level_init[7] =
{
&bc7e_scalar::bc7e_compress_block_params_init_ultrafast, // 0
&bc7e_scalar::bc7e_compress_block_params_init_veryfast, // 1
&bc7e_scalar::bc7e_compress_block_params_init_fast, // 2
&bc7e_scalar::bc7e_compress_block_params_init_basic, // 3
&bc7e_scalar::bc7e_compress_block_params_init_slow, // 4
&bc7e_scalar::bc7e_compress_block_params_init_veryslow, // 5
&bc7e_scalar::bc7e_compress_block_params_init_slowest, // 6
};
const int lvl = clamp<int>(params.m_bc7e_scalar_level, 0, 6);
// Mirrors the XUBC7 path: for perceptual (sRGB) sources run bc7e in its built-in perceptual error
// mode (it ignores m_weights then); for linear sources run linear and hand it the same RGBA channel
// weights XUASTC LDR / XUBC7 use.
const bool perceptual = comp.get_params().m_perceptual;
s_bc7e_level_init[lvl](&bc7ctx.m_bc7e_params, perceptual);
if (!perceptual)
{
for (uint32_t i = 0; i < 4; i++)
bc7ctx.m_bc7e_params.m_weights[i] = comp.get_params().m_xuastc_ldr_channel_weights[i];
}
}
// Optional debug logging (gated by the compressor's m_debug param, same flag -debug/-verbose set).
const bool debug = comp.get_params().m_debug;
if (debug)
{
const bool wrote_srgb_dbg = want_srgb && (info.m_dxgi_srgb != 0);
debug_printf("DDS export: format \"%s\" (%s), DXGI=%u, %ux%u, %u level(s), %u layer(s), %u face(s) (%s), %u subresource(s)\n",
info.m_pToken, wrote_srgb_dbg ? "sRGB" : "UNORM", dxgi_format, base_width, base_height,
total_levels, total_layers, total_faces, is_cubemap ? "cubemap" : "2D", total_subresources);
debug_printf("DDS export: %s, %u byte(s) per %s\n",
info.m_block_compressed ? "block-compressed" : "uncompressed", info.m_bytes,
info.m_block_compressed ? "block" : "pixel");
if (fmt == cDDSFmtBC7)
{
// Print the FULL BC7 configuration (both encoders' settings) regardless of which encoder
// is active, so the actual values used can be verified at a glance. The "encoder=" line
// states which one is in effect; the line for the other encoder is informational.
const bool using_bc7e = (params.m_bc7_encoder == cDDSBC7Encoder_BC7E_Scalar);
const int bc7f_lvl = clamp<int>(params.m_bc7f_level, cDDSBC7FLevel_Analytical, cDDSBC7FLevel_NonAnalytical);
const char* pBc7fLvl = (bc7f_lvl == cDDSBC7FLevel_Analytical) ? "analytical" :
(bc7f_lvl == cDDSBC7FLevel_NonAnalytical) ? "non-analytical" : "partially-analytical";
const int bc7e_lvl = clamp<int>(params.m_bc7e_scalar_level, 0, 6);
const bool perceptual = comp.get_params().m_perceptual;
const uint32_t* pW = comp.get_params().m_xuastc_ldr_channel_weights;
debug_printf("DDS export: BC7 encoder=%s\n", using_bc7e ? "bc7e_scalar" : "bc7f");
debug_printf("DDS export: bc7f level=%d (%s), pack flags=0x%X%s\n",
bc7f_lvl, pBc7fLvl, bc7ctx.m_bc7f_flags, using_bc7e ? " (inactive)" : " (active)");
debug_printf("DDS export: bc7e_scalar level=%d, perceptual=%u, weights=[%u %u %u %u]%s%s\n",
bc7e_lvl, (uint32_t)perceptual, pW[0], pW[1], pW[2], pW[3],
perceptual ? " (weights ignored in perceptual mode)" : "",
using_bc7e ? " (active)" : " (inactive)");
}
}
// Assemble the file into the caller's buffer: header then subresources in DDS order (layer, face, mip).
build_dx10_header(dds_data, base_width, base_height, total_levels, total_layers, is_cubemap, dxgi_format, info);
if (comp.get_params().m_status_output)
printf("Writing DDS (format \"%s\", %ux%u): %u subresource(s) [%u level(s), %u layer(s), %u face(s)]\n",
info.m_pToken, base_width, base_height, total_subresources, total_levels, total_layers, total_faces);
for (uint32_t layer = 0; layer < total_layers; layer++)
{
for (uint32_t face = 0; face < total_faces; face++)
{
for (uint32_t level = 0; level < total_levels; level++)
{
const uint32_t map_index = (layer * total_faces + face) * total_levels + level;
const int slice_index = slice_map[map_index];
const basisu_backend_slice_desc& desc = descs[slice_index];
if (comp.get_params().m_status_output)
printf("DDS: packing subresource %u/%u (layer %u, face %u, level %u): %ux%u\n",
map_index + 1, total_subresources, layer, face, level, desc.m_orig_width, desc.m_orig_height);
const size_t before_size = dds_data.size();
pack_slice(dds_data, slices[slice_index], desc.m_orig_width, desc.m_orig_height, fmt, bc7ctx);
if (debug)
debug_printf(" subresource [layer %u, face %u, level %u]: %ux%u -> %u byte(s)\n",
layer, face, level, desc.m_orig_width, desc.m_orig_height, (uint32_t)(dds_data.size() - before_size));
}
}
}
if (pOut_width) *pOut_width = base_width;
if (pOut_height) *pOut_height = base_height;
if (pOut_levels) *pOut_levels = total_levels;
if (pOut_layers) *pOut_layers = total_layers;
if (pOut_faces) *pOut_faces = total_faces;
return true;
}
} // namespace basisu

123
encoder/basisu_dds_export.h Normal file
View File

@@ -0,0 +1,123 @@
// basisu_dds_export.h
// Copyright (C) 2019-2025 Binomial LLC. All Rights Reserved.
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//
// Basic DX10-style .DDS writer for basisu_tool's -dds command. It consumes the prepared source image
// slices from a basis_compressor that has been run in process_source_images() mode (so all the image
// loading, mipmap generation, and cubemap/array layout is done by the encoder), then packs each slice
// into the requested output format and writes a modern DX10 (DDS_HEADER_DXT10) .dds file.
//
// Supported -dds_format tokens (see parse_dds_output_format):
// Block compressed: bc1 bc2 bc3 bc4 bc5 bc7
// Uncompressed: a8r8g8b8 r8 r8g8 r5g6b5 a1r5g5b5 a4r4g4b4
//
// The sRGB-vs-UNORM DXGI variant is selected from the compressor's transfer function flag
// (m_ktx2_and_basis_srgb_transfer_function, set by -srgb/-photo/-linear); only bc1/bc2/bc3/bc7/a8r8g8b8/a8b8g8r8
// have an sRGB DXGI variant, the rest are always written as UNORM. No transfer-function conversion is
// ever applied to the texel data - we only pick the format tag and pack the bytes as-is.
#pragma once
#include <string>
#include "../transcoder/basisu.h" // for basisu::uint8_vec
namespace basisu
{
class basis_compressor;
// The output formats supported by the -dds command.
enum dds_output_format
{
cDDSFmtInvalid = -1,
// Block compressed (packed with the real-time BC encoders).
cDDSFmtBC1,
cDDSFmtBC2,
cDDSFmtBC3,
cDDSFmtBC4,
cDDSFmtBC5,
cDDSFmtBC7,
// Uncompressed (simple pixel format conversion).
cDDSFmtA8R8G8B8, // DXGI B8G8R8A8 (32bpp, BGRA in memory), has sRGB variant
cDDSFmtA8B8G8R8, // DXGI R8G8B8A8 (32bpp, RGBA in memory), has sRGB variant
cDDSFmtR8, // DXGI R8 (8bpp)
cDDSFmtR8G8, // DXGI R8G8 (16bpp), source R->R, source G->G (like BC5)
cDDSFmtR5G6B5, // DXGI B5G6R5 (16bpp)
cDDSFmtA1R5G5B5, // DXGI B5G5R5A1 (16bpp)
cDDSFmtA4R4G4B4, // DXGI B4G4R4A4 (16bpp)
cDDSFmtTotal
};
// Parses a -dds_format token (case-insensitive) into a dds_output_format. Returns false (and sets
// fmt to cDDSFmtInvalid) on an unrecognized token.
bool parse_dds_output_format(const char* pToken, dds_output_format& fmt);
// Returns the canonical token string for a format (for messages), or "?" if invalid.
const char* get_dds_output_format_string(dds_output_format fmt);
// True if the format has a distinct sRGB DXGI variant (bc1/bc3/bc7/a8r8g8b8); false for the formats
// that are always written UNORM. Lets a caller report the actually-emitted variant accurately.
bool dds_output_format_has_srgb_variant(dds_output_format fmt);
// Which BC7 base packer to use for bc7 output. (Eventually a third encoder may be added.)
enum dds_bc7_encoder
{
cDDSBC7Encoder_Default = -1, // "not specified" sentinel (CLI use); resolves to the dds_export_params default
cDDSBC7Encoder_BC7F = 0, // built-in fast real-time packer (basist::bc7f), default
cDDSBC7Encoder_BC7E_Scalar = 1 // higher-quality (slower) scalar bc7e encoder
};
// bc7f quality level -> which analytical mode flags the bc7f packer uses (higher = slower/better).
enum dds_bc7f_level
{
cDDSBC7FLevel_Analytical = 0, // cPackBC7FlagDefault
cDDSBC7FLevel_PartiallyAnalytical = 1, // cPackBC7FlagDefaultPartiallyAnalytical (default)
cDDSBC7FLevel_NonAnalytical = 2 // cPackBC7FlagDefaultNonAnalytical (slowest/best)
};
// Input parameters for build_dds(). Intended to grow as more -dds options are added (BC1 quality, etc.).
struct dds_export_params
{
dds_output_format m_format;
// BC7-only knobs (ignored for non-bc7 formats):
dds_bc7_encoder m_bc7_encoder; // which BC7 base packer
int m_bc7f_level; // dds_bc7f_level [0,2]: analytical / partially / non analytical (bc7f only)
int m_bc7e_scalar_level; // bc7e_scalar quality level [0,6], 0=ultrafast..6=slowest (bc7e_scalar only)
dds_export_params()
: m_format(cDDSFmtInvalid),
m_bc7_encoder(cDDSBC7Encoder_BC7F),
m_bc7f_level(cDDSBC7FLevel_PartiallyAnalytical),
m_bc7e_scalar_level(2)
{ }
explicit dds_export_params(dds_output_format format)
: m_format(format),
m_bc7_encoder(cDDSBC7Encoder_BC7F),
m_bc7f_level(cDDSBC7FLevel_PartiallyAnalytical),
m_bc7e_scalar_level(2)
{ }
};
// Builds an in-memory DX10 .dds file from the prepared slices of comp (which must have had a successful
// process_source_images() call) into dds_data (which is resize(0)'d first, then appended to). The caller
// is responsible for writing dds_data out (or using it however it likes). Picks the UNORM vs sRGB DXGI
// variant from comp's transfer function flag. On failure returns false and sets error_msg. On success,
// the optional out_* values report the texture's dimensions/structure (for a caller summary).
bool build_dds(uint8_vec& dds_data, const basis_compressor& comp, const dds_export_params& params,
std::string& error_msg,
uint32_t* pOut_width = nullptr, uint32_t* pOut_height = nullptr,
uint32_t* pOut_levels = nullptr, uint32_t* pOut_layers = nullptr, uint32_t* pOut_faces = nullptr);
} // namespace basisu

File diff suppressed because it is too large Load Diff

View File

@@ -0,0 +1,200 @@
// File: basisu_xbc7_encode.h
// XBC7: experimental supercompressed-BC7 codec (prototype moved out of basisu_tool.cpp).
#pragma once
#include "basisu_enc.h"
#include "../transcoder/basisu_transcoder_internal.h"
namespace basisu {
namespace xbc7 {
const uint32_t DEFAULT_EFFORT_LEVEL = 3;
// Default quality "level" for the optional bc7e_scalar BC7 base encoder.
const uint32_t DEFAULT_BC7E_SCALAR_LEVEL = 2;
// Supported bc7e_scalar quality level range (higher = slower/better). Maps to the
// encoder's ultrafast..slowest presets where it's invoked; clamp to this range.
const uint32_t BC7E_SCALAR_MIN_LEVEL = 0;
const uint32_t BC7E_SCALAR_MAX_LEVEL = 6;
// Which BC7 base encoder XBC7 uses to pack the initial BC7 blocks before
// supercompression. cBC7F is the built-in fast real-time packer (the default);
// cBC7E_Scalar is the higher-quality (slower) scalar bc7e encoder.
enum class bc7_encoder_type
{
cBC7F = 0,
cBC7E_Scalar = 1
};
struct pack_options
{
uint32_t m_dct_q = 100; // [1,100]; 100 == lossless mode (weights always residual DPCM)
uint32_t m_weights[4] = { 1, 1, 1, 1 };
// Encoder effort/speed knob [0,10]: 0 == fastest, 10 == slowest/best
// (9 is the usual practical max; 10 exists for very high core counts).
// Currently controls how many weight-grid predictors the per-block DCT and
// DPCM sweeps evaluate: at 10 the full candidate set is searched (identical
// output to before this knob existed); as effort drops the sweep is pruned
// down to a small core of empirically high-value predictors (~6 at effort
// 0), trading a little ratio for a large drop in encode time -- the DCT
// predictor search (Q < 100) is the dominant cost. Default
// DEFAULT_EFFORT_LEVEL (a balanced speed/quality point, NOT the full
// search); pass 10 to reproduce the pre-knob output exactly.
uint32_t m_effort_level = DEFAULT_EFFORT_LEVEL;
// Optimize weights after encoding with bc7f. bc7f is a fast real-time BC7
// packer whose per-texel weights aren't optimal; when true, every block's
// weights are exhaustively re-derived for its (unchanged) config+endpoints
// right after the base pack. Slower (but threaded with the base pack);
// endpoints/config are untouched, so quality only holds or improves.
// Default off.
bool m_optimize_weights_after_bc7f = false;
// Flags for the underlying real-time BC7 base packer
// (basist::bc7f::fast_pack_bc7_auto_rgba). See bc7f::cPackBC7Flag* /
// cPackBC7FlagDefault* in basisu_transcoder_internal.h -- controls the
// quality/speed of the BC7 blocks XBC7 then supercompresses.
uint32_t m_bc7_pack_flags = basist::bc7f::cPackBC7FlagDefaultPartiallyAnalytical;
// Selects which BC7 base encoder packs the initial blocks. Default cBC7F
// (the existing fast built-in packer); cBC7E_Scalar selects the higher-quality
// scalar bc7e encoder at m_bc7e_scalar_level. Affects only the primary base
// pack -- the optional alt-pack RDO path still uses bc7f for now.
bc7_encoder_type m_bc7_encoder = bc7_encoder_type::cBC7F;
// Quality level for the bc7e_scalar encoder (only used when m_bc7_encoder ==
// cBC7E_Scalar). Higher = slower/better. Default DEFAULT_BC7E_SCALAR_LEVEL (2).
uint32_t m_bc7e_scalar_level = DEFAULT_BC7E_SCALAR_LEVEL;
// True if the source is perceptual (sRGB) data. Currently only consulted by
// the bc7e_scalar base encoder: when true it runs in its built-in perceptual
// error mode (and ignores m_weights); when false it runs linear and honors
// m_weights. Mirror basis_compressor_params::m_perceptual here.
bool m_perceptual = false;
// When true (the default), the encoder decodes the stream it just produced
// (via the transcoder) and verifies every logical BC7 block round-trips
// exactly to the coded blocks; a failure fails the encode. Guarantees every
// emitted stream is valid/decodable. Disable only to skip the extra decode.
bool m_self_validate = true;
// Optional "poor man's RDO" on the BC7 base pack. When enabled, every
// block is ALSO packed with m_bc7_pack_flags_alt (typically a cheaper,
// e.g. mode-6-only set). Both candidates are unpacked and their RGBA PSNR
// vs the source measured; the cheaper alternate is KEPT as long as it's no
// more than m_bc7_alt_max_psnr_drop dB worse than the primary. Trades a
// little quality for fewer command/config bits downstream. Default off.
bool m_bc7_alt_pack_enabled = false;
uint32_t m_bc7_pack_flags_alt = basist::bc7f::cPackBC7FlagPBitOpt | basist::bc7f::cPackBC7FlagUseTrivialMode6;
float m_bc7_alt_max_psnr_drop = 0.5f; // dB the alternate may be worse than the primary and still be kept
// Optional block-reuse RDO pre-pass (runs after the BC7 base pack, BEFORE
// the endpoint RDO -- once a whole block is reused there's no point trying
// cheaper endpoints). For each non-solid block it tries replacing it with:
// - REPEAT: an exact copy of its left/upper causal neighbor (the cheapest
// command -- one byte, no config/endpoints/weights), kept if within
// m_repeat_rdo_max_psnr_drop dB; and/or
// - SOLID: its mean color (cheap SolidDPCM command), kept if within
// m_solid_rdo_max_psnr_drop dB.
// Repeat is preferred over Solid (cheaper). Each independently enabled; off
// by default.
bool m_repeat_rdo_enabled = false;
float m_repeat_rdo_max_psnr_drop = 0.5f; // dB a block may drop to become a Repeat of a neighbor
bool m_solid_rdo_enabled = false;
float m_solid_rdo_max_psnr_drop = 0.5f; // dB a block may drop to become a solid-color block
// Optional endpoint-prediction RDO pre-pass (runs after the BC7 base pack,
// before stripe coding). For each block it tries forcing the endpoints to
// each valid causal neighbor's prediction (left/upper/left-diag/right-diag)
// via a zero-residual endpoint DPCM and keeps the best whose weighted RGBA
// PSNR drops by no more than m_endpoint_rdo_max_psnr_drop dB -- those zero
// residuals then cost almost nothing downstream. Default off.
bool m_endpoint_rdo_enabled = false;
float m_endpoint_rdo_max_psnr_drop = 0.5f; // dB the block PSNR may drop to adopt a neighbor's endpoints
// Optional weight-DCT AC-truncation RDO (DCT-coded blocks only). After a
// block is DCT-coded, its highest-frequency weight-DCT AC coefficients are
// zeroed one at a time in reverse zigzag order (the 2x2 low-freq corner DC,
// (1,0),(0,1),(1,1) is protected) while the decoded PSNR stays within this
// dB drop -- fewer coded ACs. 0 == disabled.
float m_ac_trunc_rdo_max_psnr_drop = 0.0f;
// Quality floor shared by ALL RDO passes (block-reuse + endpoint + alt-pack):
// a candidate whose absolute weighted RGBA PSNR is below this is rejected
// outright, even if its drop is within tolerance -- prevents accepting
// excessive distortion in already-degraded blocks.
float m_rdo_min_block_psnr = 33.0f; // dB
// Master switch for ALL development/debug console output (printed via
// fmt_debug_printf, so it also respects the global enable_debug_printf()
// and stays silent in normal builds). Default false. Nothing prints
// unless this is true.
bool m_debug_output = false;
// Print the detailed pack statistics dashboard. Requires m_debug_output
// to also be set (m_debug_output gates everything).
bool m_print_stats = false;
// When true, the encoder writes visualization PNG(s) -- one pixel block
// per BC7 block -- to (m_debug_file_prefix + "<name>.png"). Encode-time
// only; never affects the compressed output. Default off.
bool m_debug_images = false;
std::string m_debug_file_prefix;
// Weight DCT-vs-DPCM decision: choose lossless DPCM when
// dpcm_cost * 100 <= dct_cost * alpha. Both costs are measured-rate
// estimates (see g_dpcm_resid_cost_obits + the DCT byte cost
// constants), so 100 is the neutral operating point; raising alpha
// flips more blocks to lossless DPCM (quality up, size up).
uint32_t m_wt_dpcm_alpha_pct = 100;
// REQUIRED, never null: the caller's job pool (total threads INCLUDES
// the caller, basisu convention). The encoder always uses it; a pool of 1
// just runs serially. Pool size affects scheduling only, never the emitted
// bytes (the stripe count is independent of it).
job_pool* m_pJob_pool = nullptr;
// Stripe count for the main coding pass: 0 == auto (derived from image
// dimensions); else force exactly this many, clamped to
// [1, min(block_rows, XBC7_MAX_ENCODER_STRIPES)]. 1 == single-stripe (max
// ratio: no seam cost, no seek table). More stripes == more decode
// parallelism but a slightly larger file (per-seam prediction reset +
// seek-table cost); more stripes than pool threads simply queue. Prefer
// set_num_stripes_for_image() over assigning this directly.
uint32_t m_num_stripes = 0;
// Validate a desired stripe count against an image and store it in
// m_num_stripes. Clamps so every stripe holds at least the minimum
// efficient number of block rows (never tiny or empty stripes) and never
// exceeds the format max; images too short to stripe usefully collapse to
// a single stripe. Returns the actual count stored (>= 1).
uint32_t set_num_stripes_for_image(const image& img, uint32_t desired_num_stripes);
// Configure ALL "poor man's RDO" knobs from a single level [0,100].
// 0 -> RDO fully OFF: every enable flag false, every PSNR-drop 0.
// [1,100] -> enables the RDO passes and linearly ramps their tolerated
// per-block PSNR drop with the level: the general passes
// (BC7 alt-pack, endpoint-DPCM, weight-DCT AC-truncation) up to
// 10 dB at 100, and the cheaper block-reuse passes (repeat,
// solid) up to 4 dB at 100.
// The absolute quality floor (m_rdo_min_block_psnr) is left untouched -- it
// still gates every candidate regardless of level.
void set_rdo_level(uint32_t rdo_level);
};
// coded_log_blocks (REQUIRED, by reference): receives the final coded logical
// BC7 blocks (resized + filled by the encoder). It's mandatory because the
// encoder always decodes the stream it just produced and validates every block
// round-trips to these -- see the self-validation at the end of pack_image().
bool pack_image(
const image& orig_img,
const pack_options& opts,
uint8_vec& comp_bytes,
vector2D<basist::bc7u::log_bc7_block>& coded_log_blocks);
// NOTE: the decoder API (unpack_image / unpack_image_threaded) now lives in
// basisu_xbc7_decode.h.
} // namespace xbc7
} // namespace basisu