VBO updates from Vir Lindens shining fixes. VBO mapping perf improvement. Alpha rigged attachments render fix, hopefully. Crashfix in void pushWireframe.
This commit is contained in:
@@ -695,6 +695,49 @@ static void xform(LLVector2 &tex_coord, F32 cosAng, F32 sinAng, F32 offS, F32 of
|
||||
tex_coord.mV[1] = t;
|
||||
}
|
||||
|
||||
// Transform the texture coordinates for this face.
|
||||
static void xform4a(LLVector4a &tex_coord, const LLVector4a& trans, const LLVector4Logical& mask, const LLVector4a& rot0, const LLVector4a& rot1, const LLVector4a& offset, const LLVector4a& scale)
|
||||
{
|
||||
//tex coord is two coords, <s0, t0, s1, t1>
|
||||
LLVector4a st;
|
||||
|
||||
// Texture transforms are done about the center of the face.
|
||||
st.setAdd(tex_coord, trans);
|
||||
|
||||
// Handle rotation
|
||||
LLVector4a rot_st;
|
||||
|
||||
// <s0 * cosAng, s0*-sinAng, s1*cosAng, s1*-sinAng>
|
||||
LLVector4a s0;
|
||||
s0.splat(st, 0);
|
||||
LLVector4a s1;
|
||||
s1.splat(st, 2);
|
||||
LLVector4a ss;
|
||||
ss.setSelectWithMask(mask, s1, s0);
|
||||
|
||||
LLVector4a a;
|
||||
a.setMul(rot0, ss);
|
||||
|
||||
// <t0*sinAng, t0*cosAng, t1*sinAng, t1*cosAng>
|
||||
LLVector4a t0;
|
||||
t0.splat(st, 1);
|
||||
LLVector4a t1;
|
||||
t1.splat(st, 3);
|
||||
LLVector4a tt;
|
||||
tt.setSelectWithMask(mask, t1, t0);
|
||||
|
||||
LLVector4a b;
|
||||
b.setMul(rot1, tt);
|
||||
|
||||
st.setAdd(a,b);
|
||||
|
||||
// Then scale
|
||||
st.mul(scale);
|
||||
|
||||
// Then offset
|
||||
tex_coord.setAdd(st, offset);
|
||||
}
|
||||
|
||||
|
||||
bool less_than_max_mag(const LLVector4a& vec)
|
||||
{
|
||||
@@ -1069,7 +1112,9 @@ BOOL LLFace::getGeometryVolume(const LLVolume& volume,
|
||||
updateRebuildFlags();
|
||||
}
|
||||
|
||||
bool map_range = gGLManager.mHasMapBufferRange || gGLManager.mHasFlushBufferRange;
|
||||
|
||||
//don't use map range (generates many redundant unmap calls)
|
||||
bool map_range = false; //gGLManager.mHasMapBufferRange || gGLManager.mHasFlushBufferRange;
|
||||
|
||||
if (mVertexBuffer.notNull())
|
||||
{
|
||||
@@ -1095,16 +1140,12 @@ BOOL LLFace::getGeometryVolume(const LLVolume& volume,
|
||||
}
|
||||
|
||||
LLStrider<LLVector3> vert;
|
||||
LLVector4a* vertices = NULL;
|
||||
LLStrider<LLVector2> tex_coords;
|
||||
LLStrider<LLVector2> tex_coords2;
|
||||
LLVector4a* normals = NULL;
|
||||
LLStrider<LLVector3> norm;
|
||||
LLStrider<LLColor4U> colors;
|
||||
LLVector4a* binormals = NULL;
|
||||
LLStrider<LLVector3> binorm;
|
||||
LLStrider<U16> indicesp;
|
||||
LLVector4a* weights = NULL;
|
||||
LLStrider<LLVector4> wght;
|
||||
|
||||
BOOL full_rebuild = force_rebuild || mDrawablep->isState(LLDrawable::REBUILD_VOLUME);
|
||||
@@ -1172,7 +1213,7 @@ BOOL LLFace::getGeometryVolume(const LLVolume& volume,
|
||||
LLFastTimer t(LLFastTimer::FTM_FACE_GEOM_INDEX);
|
||||
mVertexBuffer->getIndexStrider(indicesp, mIndicesIndex, mIndicesCount, map_range);
|
||||
|
||||
__m128i* dst = (__m128i*) indicesp.get();
|
||||
volatile __m128i* dst = (__m128i*) indicesp.get();
|
||||
__m128i* src = (__m128i*) vf.mIndices;
|
||||
__m128i offset = _mm_set1_epi16(index_offset);
|
||||
|
||||
@@ -1181,12 +1222,17 @@ BOOL LLFace::getGeometryVolume(const LLVolume& volume,
|
||||
for (S32 i = 0; i < end; i++)
|
||||
{
|
||||
__m128i res = _mm_add_epi16(src[i], offset);
|
||||
_mm_storeu_si128(dst+i, res);
|
||||
_mm_storeu_si128((__m128i*) dst++, res);
|
||||
}
|
||||
|
||||
for (S32 i = end*8; i < num_indices; ++i)
|
||||
{
|
||||
indicesp[i] = vf.mIndices[i]+index_offset;
|
||||
//LLFastTimer t(LLFastTimer::FTM_FACE_GEOM_INDEX_TAIL);
|
||||
U16* idx = (U16*) dst;
|
||||
|
||||
for (S32 i = end*8; i < num_indices; ++i)
|
||||
{
|
||||
*idx++ = vf.mIndices[i]+index_offset;
|
||||
}
|
||||
}
|
||||
|
||||
if (map_range)
|
||||
@@ -1351,11 +1397,37 @@ BOOL LLFace::getGeometryVolume(const LLVolume& volume,
|
||||
}
|
||||
else
|
||||
{
|
||||
for (S32 i = 0; i < num_vertices; i++)
|
||||
F32* dst = (F32*) tex_coords.get();
|
||||
LLVector4a* src = (LLVector4a*) vf.mTexCoords;
|
||||
|
||||
LLVector4a trans;
|
||||
trans.splat(-0.5f);
|
||||
|
||||
LLVector4a rot0;
|
||||
rot0.set(cos_ang, -sin_ang, cos_ang, -sin_ang);
|
||||
|
||||
LLVector4a rot1;
|
||||
rot1.set(sin_ang, cos_ang, sin_ang, cos_ang);
|
||||
|
||||
LLVector4a scale;
|
||||
scale.set(ms, mt, ms, mt);
|
||||
|
||||
LLVector4a offset;
|
||||
offset.set(os+0.5f, ot+0.5f, os+0.5f, ot+0.5f);
|
||||
|
||||
LLVector4Logical mask;
|
||||
mask.clear();
|
||||
mask.setElement<2>();
|
||||
mask.setElement<3>();
|
||||
|
||||
U32 count = num_vertices/2 + num_vertices%2;
|
||||
|
||||
for (U32 i = 0; i < count; i++)
|
||||
{
|
||||
LLVector2 tc(vf.mTexCoords[i]);
|
||||
xform(tc, cos_ang, sin_ang, os, ot, ms, mt);
|
||||
*tex_coords++ = tc;
|
||||
LLVector4a res = *src++;
|
||||
xform4a(res, trans, mask, rot0, rot1, offset, scale);
|
||||
res.store4a(dst);
|
||||
dst += 4;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1526,44 +1598,53 @@ BOOL LLFace::getGeometryVolume(const LLVolume& volume,
|
||||
llassert(num_vertices > 0);
|
||||
|
||||
mVertexBuffer->getVertexStrider(vert, mGeomIndex, mGeomCount, map_range);
|
||||
vertices = (LLVector4a*) vert.get();
|
||||
|
||||
|
||||
|
||||
LLMatrix4a mat_vert;
|
||||
mat_vert.loadu(mat_vert_in);
|
||||
|
||||
LLVector4a* src = vf.mPositions;
|
||||
LLVector4a* dst = vertices;
|
||||
volatile F32* dst = (volatile F32*) vert.get();
|
||||
|
||||
LLVector4a* end = dst+num_vertices;
|
||||
do
|
||||
{
|
||||
mat_vert.affineTransform(*src++, *dst++);
|
||||
}
|
||||
while(dst < end);
|
||||
volatile F32* end = dst+num_vertices*4;
|
||||
LLVector4a res;
|
||||
|
||||
LLVector4a texIdx;
|
||||
|
||||
F32 index = (F32) (mTextureIndex < 255 ? mTextureIndex : 0);
|
||||
|
||||
llassert(index <= LLGLSLShader::sIndexedTextureChannels-1);
|
||||
F32 *index_dst = (F32*) vertices;
|
||||
F32 *index_end = (F32*) end;
|
||||
|
||||
index_dst += 3;
|
||||
index_end += 3;
|
||||
do
|
||||
{
|
||||
*index_dst = index;
|
||||
index_dst += 4;
|
||||
}
|
||||
while (index_dst < index_end);
|
||||
LLVector4Logical mask;
|
||||
mask.clear();
|
||||
mask.setElement<3>();
|
||||
|
||||
S32 aligned_pad_vertices = mGeomCount - num_vertices;
|
||||
LLVector4a* last_vec = end - 1;
|
||||
while (aligned_pad_vertices > 0)
|
||||
texIdx.set(0,0,0,index);
|
||||
|
||||
{
|
||||
--aligned_pad_vertices;
|
||||
*dst++ = *last_vec;
|
||||
LLVector4a tmp;
|
||||
|
||||
do
|
||||
{
|
||||
mat_vert.affineTransform(*src++, res);
|
||||
tmp.setSelectWithMask(mask, texIdx, res);
|
||||
tmp.store4a((F32*) dst);
|
||||
dst += 4;
|
||||
}
|
||||
while(dst < end);
|
||||
}
|
||||
|
||||
|
||||
{
|
||||
S32 aligned_pad_vertices = mGeomCount - num_vertices;
|
||||
res.set(res[0], res[1], res[2], 0.f);
|
||||
|
||||
while (aligned_pad_vertices > 0)
|
||||
{
|
||||
--aligned_pad_vertices;
|
||||
res.store4a((F32*) dst);
|
||||
dst += 4;
|
||||
}
|
||||
}
|
||||
|
||||
if (map_range)
|
||||
{
|
||||
mVertexBuffer->flush();
|
||||
@@ -1574,14 +1655,15 @@ BOOL LLFace::getGeometryVolume(const LLVolume& volume,
|
||||
{
|
||||
LLFastTimer t(LLFastTimer::FTM_FACE_GEOM_NORMAL);
|
||||
mVertexBuffer->getNormalStrider(norm, mGeomIndex, mGeomCount, map_range);
|
||||
normals = (LLVector4a*) norm.get();
|
||||
F32* normals = (F32*) norm.get();
|
||||
|
||||
for (S32 i = 0; i < num_vertices; i++)
|
||||
{
|
||||
LLVector4a normal;
|
||||
mat_normal.rotate(vf.mNormals[i], normal);
|
||||
normal.normalize3fast();
|
||||
normals[i] = normal;
|
||||
normal.store4a(normals);
|
||||
normals += 4;
|
||||
}
|
||||
|
||||
if (map_range)
|
||||
@@ -1594,14 +1676,15 @@ BOOL LLFace::getGeometryVolume(const LLVolume& volume,
|
||||
{
|
||||
LLFastTimer t(LLFastTimer::FTM_FACE_GEOM_BINORMAL);
|
||||
mVertexBuffer->getBinormalStrider(binorm, mGeomIndex, mGeomCount, map_range);
|
||||
binormals = (LLVector4a*) binorm.get();
|
||||
F32* binormals = (F32*) binorm.get();
|
||||
|
||||
for (S32 i = 0; i < num_vertices; i++)
|
||||
{
|
||||
LLVector4a binormal;
|
||||
mat_normal.rotate(vf.mBinormals[i], binormal);
|
||||
binormal.normalize3fast();
|
||||
binormals[i] = binormal;
|
||||
binormal.store4a(binormals);
|
||||
binormals += 4;
|
||||
}
|
||||
|
||||
if (map_range)
|
||||
@@ -1614,8 +1697,8 @@ BOOL LLFace::getGeometryVolume(const LLVolume& volume,
|
||||
{
|
||||
LLFastTimer t(LLFastTimer::FTM_FACE_GEOM_WEIGHTS);
|
||||
mVertexBuffer->getWeight4Strider(wght, mGeomIndex, mGeomCount, map_range);
|
||||
weights = (LLVector4a*) wght.get();
|
||||
LLVector4a::memcpyNonAliased16((F32*) weights, (F32*) vf.mWeights, num_vertices*4*sizeof(F32));
|
||||
F32* weights = (F32*) wght.get();
|
||||
LLVector4a::memcpyNonAliased16(weights, (F32*) vf.mWeights, num_vertices*4*sizeof(F32));
|
||||
if (map_range)
|
||||
{
|
||||
mVertexBuffer->flush();
|
||||
@@ -1634,7 +1717,7 @@ BOOL LLFace::getGeometryVolume(const LLVolume& volume,
|
||||
|
||||
src.loadua((F32*) vec);
|
||||
|
||||
LLVector4a* dst = (LLVector4a*) colors.get();
|
||||
F32* dst = (F32*) colors.get();
|
||||
S32 num_vecs = num_vertices/4;
|
||||
if (num_vertices%4 > 0)
|
||||
{
|
||||
@@ -1643,7 +1726,8 @@ BOOL LLFace::getGeometryVolume(const LLVolume& volume,
|
||||
|
||||
for (S32 i = 0; i < num_vecs; i++)
|
||||
{
|
||||
dst[i] = src;
|
||||
src.store4a(dst);
|
||||
dst += 4;
|
||||
}
|
||||
|
||||
if (map_range)
|
||||
@@ -1673,7 +1757,7 @@ BOOL LLFace::getGeometryVolume(const LLVolume& volume,
|
||||
|
||||
src.loadua((F32*) vec);
|
||||
|
||||
LLVector4a* dst = (LLVector4a*) emissive.get();
|
||||
F32* dst = (F32*) emissive.get();
|
||||
S32 num_vecs = num_vertices/4;
|
||||
if (num_vertices%4 > 0)
|
||||
{
|
||||
@@ -1682,7 +1766,8 @@ BOOL LLFace::getGeometryVolume(const LLVolume& volume,
|
||||
|
||||
for (S32 i = 0; i < num_vecs; i++)
|
||||
{
|
||||
dst[i] = src;
|
||||
src.store4a(dst);
|
||||
dst += 4;
|
||||
}
|
||||
|
||||
if (map_range)
|
||||
|
||||
Reference in New Issue
Block a user