mirror of
https://github.com/bkaradzic/bgfx.git
synced 2026-08-30 15:08:31 +00:00
Updated meshoptimizer.
This commit is contained in:
206
3rdparty/meshoptimizer/src/spatialorder.cpp
vendored
206
3rdparty/meshoptimizer/src/spatialorder.cpp
vendored
@@ -22,7 +22,7 @@ inline unsigned long long part1By2(unsigned long long x)
|
||||
return x;
|
||||
}
|
||||
|
||||
static void computeOrder(unsigned long long* result, const float* vertex_positions_data, size_t vertex_count, size_t vertex_positions_stride)
|
||||
static void computeOrder(unsigned long long* result, const float* vertex_positions_data, size_t vertex_count, size_t vertex_positions_stride, bool morton)
|
||||
{
|
||||
size_t vertex_stride_float = vertex_positions_stride / sizeof(float);
|
||||
|
||||
@@ -60,61 +60,158 @@ static void computeOrder(unsigned long long* result, const float* vertex_positio
|
||||
int y = int((v[1] - minv[1]) * scale + 0.5f);
|
||||
int z = int((v[2] - minv[2]) * scale + 0.5f);
|
||||
|
||||
result[i] = part1By2(x) | (part1By2(y) << 1) | (part1By2(z) << 2);
|
||||
if (morton)
|
||||
result[i] = part1By2(x) | (part1By2(y) << 1) | (part1By2(z) << 2);
|
||||
else
|
||||
result[i] = ((unsigned long long)x << 0) | ((unsigned long long)y << 20) | ((unsigned long long)z << 40);
|
||||
}
|
||||
}
|
||||
|
||||
static void computeHistogram(unsigned int (&hist)[1024][5], const unsigned long long* data, size_t count)
|
||||
static void radixSort10(unsigned int* destination, const unsigned int* source, const unsigned short* keys, size_t count)
|
||||
{
|
||||
unsigned int hist[1024];
|
||||
memset(hist, 0, sizeof(hist));
|
||||
|
||||
// compute 5 10-bit histograms in parallel
|
||||
// compute histogram (assume keys are 10-bit)
|
||||
for (size_t i = 0; i < count; ++i)
|
||||
{
|
||||
unsigned long long id = data[i];
|
||||
hist[keys[i]]++;
|
||||
|
||||
hist[(id >> 0) & 1023][0]++;
|
||||
hist[(id >> 10) & 1023][1]++;
|
||||
hist[(id >> 20) & 1023][2]++;
|
||||
hist[(id >> 30) & 1023][3]++;
|
||||
hist[(id >> 40) & 1023][4]++;
|
||||
}
|
||||
|
||||
unsigned int sum0 = 0, sum1 = 0, sum2 = 0, sum3 = 0, sum4 = 0;
|
||||
unsigned int sum = 0;
|
||||
|
||||
// replace histogram data with prefix histogram sums in-place
|
||||
for (int i = 0; i < 1024; ++i)
|
||||
{
|
||||
unsigned int h0 = hist[i][0], h1 = hist[i][1], h2 = hist[i][2], h3 = hist[i][3], h4 = hist[i][4];
|
||||
unsigned int h = hist[i];
|
||||
hist[i] = sum;
|
||||
sum += h;
|
||||
}
|
||||
|
||||
assert(sum == count);
|
||||
|
||||
// reorder values
|
||||
for (size_t i = 0; i < count; ++i)
|
||||
{
|
||||
unsigned int id = keys[source[i]];
|
||||
|
||||
destination[hist[id]++] = source[i];
|
||||
}
|
||||
}
|
||||
|
||||
static void computeHistogram(unsigned int (&hist)[256][2], const unsigned short* data, size_t count)
|
||||
{
|
||||
memset(hist, 0, sizeof(hist));
|
||||
|
||||
// compute 2 8-bit histograms in parallel
|
||||
for (size_t i = 0; i < count; ++i)
|
||||
{
|
||||
unsigned long long id = data[i];
|
||||
|
||||
hist[(id >> 0) & 255][0]++;
|
||||
hist[(id >> 8) & 255][1]++;
|
||||
}
|
||||
|
||||
unsigned int sum0 = 0, sum1 = 0;
|
||||
|
||||
// replace histogram data with prefix histogram sums in-place
|
||||
for (int i = 0; i < 256; ++i)
|
||||
{
|
||||
unsigned int h0 = hist[i][0], h1 = hist[i][1];
|
||||
|
||||
hist[i][0] = sum0;
|
||||
hist[i][1] = sum1;
|
||||
hist[i][2] = sum2;
|
||||
hist[i][3] = sum3;
|
||||
hist[i][4] = sum4;
|
||||
|
||||
sum0 += h0;
|
||||
sum1 += h1;
|
||||
sum2 += h2;
|
||||
sum3 += h3;
|
||||
sum4 += h4;
|
||||
}
|
||||
|
||||
assert(sum0 == count && sum1 == count && sum2 == count && sum3 == count && sum4 == count);
|
||||
assert(sum0 == count && sum1 == count);
|
||||
}
|
||||
|
||||
static void radixPass(unsigned int* destination, const unsigned int* source, const unsigned long long* keys, size_t count, unsigned int (&hist)[1024][5], int pass)
|
||||
static void radixPass(unsigned int* destination, const unsigned int* source, const unsigned short* keys, size_t count, unsigned int (&hist)[256][2], int pass)
|
||||
{
|
||||
int bitoff = pass * 10;
|
||||
int bitoff = pass * 8;
|
||||
|
||||
for (size_t i = 0; i < count; ++i)
|
||||
{
|
||||
unsigned int id = unsigned(keys[source[i]] >> bitoff) & 1023;
|
||||
unsigned int id = unsigned(keys[source[i]] >> bitoff) & 255;
|
||||
|
||||
destination[hist[id][pass]++] = source[i];
|
||||
}
|
||||
}
|
||||
|
||||
static void partitionPoints(unsigned int* target, const unsigned int* order, const unsigned char* sides, size_t split, size_t count)
|
||||
{
|
||||
size_t l = 0, r = split;
|
||||
|
||||
for (size_t i = 0; i < count; ++i)
|
||||
{
|
||||
unsigned char side = sides[order[i]];
|
||||
target[side ? r : l] = order[i];
|
||||
l += 1;
|
||||
l -= side;
|
||||
r += side;
|
||||
}
|
||||
|
||||
assert(l == split && r == count);
|
||||
}
|
||||
|
||||
static void splitPoints(unsigned int* destination, unsigned int* orderx, unsigned int* ordery, unsigned int* orderz, const unsigned long long* keys, size_t count, void* scratch, size_t cluster_size)
|
||||
{
|
||||
if (count <= cluster_size)
|
||||
{
|
||||
memcpy(destination, orderx, count * sizeof(unsigned int));
|
||||
return;
|
||||
}
|
||||
|
||||
unsigned int* axes[3] = {orderx, ordery, orderz};
|
||||
|
||||
int bestk = -1;
|
||||
unsigned int bestdim = 0;
|
||||
|
||||
for (int k = 0; k < 3; ++k)
|
||||
{
|
||||
const unsigned int mask = (1 << 20) - 1;
|
||||
unsigned int dim = (unsigned(keys[axes[k][count - 1]] >> (k * 20)) & mask) - (unsigned(keys[axes[k][0]] >> (k * 20)) & mask);
|
||||
|
||||
if (dim >= bestdim)
|
||||
{
|
||||
bestk = k;
|
||||
bestdim = dim;
|
||||
}
|
||||
}
|
||||
|
||||
assert(bestk >= 0);
|
||||
|
||||
// split roughly in half, with the left split always being aligned to cluster size
|
||||
size_t split = ((count / 2) + cluster_size - 1) / cluster_size * cluster_size;
|
||||
assert(split > 0 && split < count);
|
||||
|
||||
// mark sides of split for partitioning
|
||||
unsigned char* sides = static_cast<unsigned char*>(scratch) + count * sizeof(unsigned int);
|
||||
|
||||
for (size_t i = 0; i < split; ++i)
|
||||
sides[axes[bestk][i]] = 0;
|
||||
|
||||
for (size_t i = split; i < count; ++i)
|
||||
sides[axes[bestk][i]] = 1;
|
||||
|
||||
// partition all axes into two sides, maintaining order
|
||||
unsigned int* temp = static_cast<unsigned int*>(scratch);
|
||||
|
||||
for (int k = 0; k < 3; ++k)
|
||||
{
|
||||
if (k == bestk)
|
||||
continue;
|
||||
|
||||
unsigned int* axis = axes[k];
|
||||
memcpy(temp, axis, sizeof(unsigned int) * count);
|
||||
partitionPoints(axis, temp, sides, split, count);
|
||||
}
|
||||
|
||||
splitPoints(destination, orderx, ordery, orderz, keys, split, scratch, cluster_size);
|
||||
splitPoints(destination + split, orderx + split, ordery + split, orderz + split, keys, count - split, scratch, cluster_size);
|
||||
}
|
||||
|
||||
} // namespace meshopt
|
||||
|
||||
void meshopt_spatialSortRemap(unsigned int* destination, const float* vertex_positions, size_t vertex_count, size_t vertex_positions_stride)
|
||||
@@ -127,22 +224,25 @@ void meshopt_spatialSortRemap(unsigned int* destination, const float* vertex_pos
|
||||
meshopt_Allocator allocator;
|
||||
|
||||
unsigned long long* keys = allocator.allocate<unsigned long long>(vertex_count);
|
||||
computeOrder(keys, vertex_positions, vertex_count, vertex_positions_stride);
|
||||
computeOrder(keys, vertex_positions, vertex_count, vertex_positions_stride, /* morton= */ true);
|
||||
|
||||
unsigned int hist[1024][5];
|
||||
computeHistogram(hist, keys, vertex_count);
|
||||
|
||||
unsigned int* scratch = allocator.allocate<unsigned int>(vertex_count);
|
||||
unsigned int* scratch = allocator.allocate<unsigned int>(vertex_count * 2); // 4b for order + 2b for keys
|
||||
unsigned short* keyk = (unsigned short*)(scratch + vertex_count);
|
||||
|
||||
for (size_t i = 0; i < vertex_count; ++i)
|
||||
destination[i] = unsigned(i);
|
||||
|
||||
unsigned int* order[] = {scratch, destination};
|
||||
|
||||
// 5-pass radix sort computes the resulting order into scratch
|
||||
radixPass(scratch, destination, keys, vertex_count, hist, 0);
|
||||
radixPass(destination, scratch, keys, vertex_count, hist, 1);
|
||||
radixPass(scratch, destination, keys, vertex_count, hist, 2);
|
||||
radixPass(destination, scratch, keys, vertex_count, hist, 3);
|
||||
radixPass(scratch, destination, keys, vertex_count, hist, 4);
|
||||
for (int k = 0; k < 5; ++k)
|
||||
{
|
||||
// copy 10-bit key segments into keyk to reduce cache pressure during radix pass
|
||||
for (size_t i = 0; i < vertex_count; ++i)
|
||||
keyk[i] = (unsigned short)((keys[i] >> (k * 10)) & 1023);
|
||||
|
||||
radixSort10(order[k % 2], order[(k + 1) % 2], keyk, vertex_count);
|
||||
}
|
||||
|
||||
// since our remap table is mapping old=>new, we need to reverse it
|
||||
for (size_t i = 0; i < vertex_count; ++i)
|
||||
@@ -202,3 +302,39 @@ void meshopt_spatialSortTriangles(unsigned int* destination, const unsigned int*
|
||||
destination[r * 3 + 2] = c;
|
||||
}
|
||||
}
|
||||
|
||||
void meshopt_spatialClusterPoints(unsigned int* destination, const float* vertex_positions, size_t vertex_count, size_t vertex_positions_stride, size_t cluster_size)
|
||||
{
|
||||
using namespace meshopt;
|
||||
|
||||
assert(vertex_positions_stride >= 12 && vertex_positions_stride <= 256);
|
||||
assert(vertex_positions_stride % sizeof(float) == 0);
|
||||
assert(cluster_size > 0);
|
||||
|
||||
meshopt_Allocator allocator;
|
||||
|
||||
unsigned long long* keys = allocator.allocate<unsigned long long>(vertex_count);
|
||||
computeOrder(keys, vertex_positions, vertex_count, vertex_positions_stride, /* morton= */ false);
|
||||
|
||||
unsigned int* order = allocator.allocate<unsigned int>(vertex_count * 3);
|
||||
unsigned int* scratch = allocator.allocate<unsigned int>(vertex_count * 2); // 4b for order + 1b for side or 2b for keys
|
||||
unsigned short* keyk = reinterpret_cast<unsigned short*>(scratch + vertex_count);
|
||||
|
||||
for (int k = 0; k < 3; ++k)
|
||||
{
|
||||
// copy 16-bit key segments into keyk to reduce cache pressure during radix pass
|
||||
for (size_t i = 0; i < vertex_count; ++i)
|
||||
keyk[i] = (unsigned short)(keys[i] >> (k * 20));
|
||||
|
||||
unsigned int hist[256][2];
|
||||
computeHistogram(hist, keyk, vertex_count);
|
||||
|
||||
for (size_t i = 0; i < vertex_count; ++i)
|
||||
order[k * vertex_count + i] = unsigned(i);
|
||||
|
||||
radixPass(scratch, order + k * vertex_count, keyk, vertex_count, hist, 0);
|
||||
radixPass(order + k * vertex_count, scratch, keyk, vertex_count, hist, 1);
|
||||
}
|
||||
|
||||
splitPoints(destination, order, order + vertex_count, order + 2 * vertex_count, keys, vertex_count, scratch, cluster_size);
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user