Vertexloader cleanup/fixing: Don't use floating point copies unless necessary, don't allow scale factors for floating point texture coordinates.
git-svn-id: https://dolphin-emu.googlecode.com/svn/trunk@1282 8ced0084-cf51-0410-be5f-012b33b47a6e
This commit is contained in:
parent
5672367b4c
commit
b19859450e
|
@ -203,9 +203,9 @@ void LOADERDECL VertexLoader_Normal::Normal_DirectShort()
|
||||||
|
|
||||||
void LOADERDECL VertexLoader_Normal::Normal_DirectFloat()
|
void LOADERDECL VertexLoader_Normal::Normal_DirectFloat()
|
||||||
{
|
{
|
||||||
((float*)VertexManager::s_pCurBufferPointer)[0] = DataReadF32();
|
((u32*)VertexManager::s_pCurBufferPointer)[0] = DataReadU32();
|
||||||
((float*)VertexManager::s_pCurBufferPointer)[1] = DataReadF32();
|
((u32*)VertexManager::s_pCurBufferPointer)[1] = DataReadU32();
|
||||||
((float*)VertexManager::s_pCurBufferPointer)[2] = DataReadF32();
|
((u32*)VertexManager::s_pCurBufferPointer)[2] = DataReadU32();
|
||||||
VertexManager::s_pCurBufferPointer += 12;
|
VertexManager::s_pCurBufferPointer += 12;
|
||||||
LOG_NORMF()
|
LOG_NORMF()
|
||||||
}
|
}
|
||||||
|
@ -238,9 +238,9 @@ void LOADERDECL VertexLoader_Normal::Normal_DirectFloat3()
|
||||||
{
|
{
|
||||||
for (int i = 0; i < 3; i++)
|
for (int i = 0; i < 3; i++)
|
||||||
{
|
{
|
||||||
((float*)VertexManager::s_pCurBufferPointer)[0] = DataReadF32();
|
((u32*)VertexManager::s_pCurBufferPointer)[0] = DataReadU32();
|
||||||
((float*)VertexManager::s_pCurBufferPointer)[1] = DataReadF32();
|
((u32*)VertexManager::s_pCurBufferPointer)[1] = DataReadU32();
|
||||||
((float*)VertexManager::s_pCurBufferPointer)[2] = DataReadF32();
|
((u32*)VertexManager::s_pCurBufferPointer)[2] = DataReadU32();
|
||||||
VertexManager::s_pCurBufferPointer += 12;
|
VertexManager::s_pCurBufferPointer += 12;
|
||||||
LOG_NORMF();
|
LOG_NORMF();
|
||||||
}
|
}
|
||||||
|
@ -279,9 +279,9 @@ void LOADERDECL VertexLoader_Normal::Normal_Index8_Float()
|
||||||
{
|
{
|
||||||
u8 Index = DataReadU8();
|
u8 Index = DataReadU8();
|
||||||
u32 iAddress = arraybases[ARRAY_NORMAL] + (Index * arraystrides[ARRAY_NORMAL]);
|
u32 iAddress = arraybases[ARRAY_NORMAL] + (Index * arraystrides[ARRAY_NORMAL]);
|
||||||
((float*)VertexManager::s_pCurBufferPointer)[0] = Memory_Read_Float(iAddress);
|
((u32*)VertexManager::s_pCurBufferPointer)[0] = Memory_Read_U32(iAddress);
|
||||||
((float*)VertexManager::s_pCurBufferPointer)[1] = Memory_Read_Float(iAddress+4);
|
((u32*)VertexManager::s_pCurBufferPointer)[1] = Memory_Read_U32(iAddress+4);
|
||||||
((float*)VertexManager::s_pCurBufferPointer)[2] = Memory_Read_Float(iAddress+8);
|
((u32*)VertexManager::s_pCurBufferPointer)[2] = Memory_Read_U32(iAddress+8);
|
||||||
VertexManager::s_pCurBufferPointer += 12;
|
VertexManager::s_pCurBufferPointer += 12;
|
||||||
LOG_NORMF();
|
LOG_NORMF();
|
||||||
}
|
}
|
||||||
|
@ -320,9 +320,9 @@ void LOADERDECL VertexLoader_Normal::Normal_Index8_Float3_Indices1()
|
||||||
for (int i = 0; i < 3; i++)
|
for (int i = 0; i < 3; i++)
|
||||||
{
|
{
|
||||||
u32 iAddress = arraybases[ARRAY_NORMAL] + (Index * arraystrides[ARRAY_NORMAL]) + 4*3*i;
|
u32 iAddress = arraybases[ARRAY_NORMAL] + (Index * arraystrides[ARRAY_NORMAL]) + 4*3*i;
|
||||||
((float*)VertexManager::s_pCurBufferPointer)[0] = Memory_Read_Float(iAddress);
|
((u32*)VertexManager::s_pCurBufferPointer)[0] = Memory_Read_U32(iAddress);
|
||||||
((float*)VertexManager::s_pCurBufferPointer)[1] = Memory_Read_Float(iAddress+4);
|
((u32*)VertexManager::s_pCurBufferPointer)[1] = Memory_Read_U32(iAddress+4);
|
||||||
((float*)VertexManager::s_pCurBufferPointer)[2] = Memory_Read_Float(iAddress+8);
|
((u32*)VertexManager::s_pCurBufferPointer)[2] = Memory_Read_U32(iAddress+8);
|
||||||
VertexManager::s_pCurBufferPointer += 12;
|
VertexManager::s_pCurBufferPointer += 12;
|
||||||
LOG_NORMF();
|
LOG_NORMF();
|
||||||
}
|
}
|
||||||
|
@ -362,9 +362,9 @@ void LOADERDECL VertexLoader_Normal::Normal_Index8_Float3_Indices3()
|
||||||
{
|
{
|
||||||
u8 Index = DataReadU8();
|
u8 Index = DataReadU8();
|
||||||
u32 iAddress = arraybases[ARRAY_NORMAL] + (Index * arraystrides[ARRAY_NORMAL]) + 4*3*i;
|
u32 iAddress = arraybases[ARRAY_NORMAL] + (Index * arraystrides[ARRAY_NORMAL]) + 4*3*i;
|
||||||
((float*)VertexManager::s_pCurBufferPointer)[0] = Memory_Read_Float(iAddress);
|
((u32*)VertexManager::s_pCurBufferPointer)[0] = Memory_Read_U32(iAddress);
|
||||||
((float*)VertexManager::s_pCurBufferPointer)[1] = Memory_Read_Float(iAddress+4);
|
((u32*)VertexManager::s_pCurBufferPointer)[1] = Memory_Read_U32(iAddress+4);
|
||||||
((float*)VertexManager::s_pCurBufferPointer)[2] = Memory_Read_Float(iAddress+8);
|
((u32*)VertexManager::s_pCurBufferPointer)[2] = Memory_Read_U32(iAddress+8);
|
||||||
VertexManager::s_pCurBufferPointer += 12;
|
VertexManager::s_pCurBufferPointer += 12;
|
||||||
LOG_NORMF();
|
LOG_NORMF();
|
||||||
}
|
}
|
||||||
|
@ -400,9 +400,9 @@ void LOADERDECL VertexLoader_Normal::Normal_Index16_Float()
|
||||||
{
|
{
|
||||||
u16 Index = DataReadU16();
|
u16 Index = DataReadU16();
|
||||||
u32 iAddress = arraybases[ARRAY_NORMAL] + (Index * arraystrides[ARRAY_NORMAL]);
|
u32 iAddress = arraybases[ARRAY_NORMAL] + (Index * arraystrides[ARRAY_NORMAL]);
|
||||||
((float*)VertexManager::s_pCurBufferPointer)[0] = Memory_Read_Float(iAddress);
|
((u32*)VertexManager::s_pCurBufferPointer)[0] = Memory_Read_U32(iAddress);
|
||||||
((float*)VertexManager::s_pCurBufferPointer)[1] = Memory_Read_Float(iAddress+4);
|
((u32*)VertexManager::s_pCurBufferPointer)[1] = Memory_Read_U32(iAddress+4);
|
||||||
((float*)VertexManager::s_pCurBufferPointer)[2] = Memory_Read_Float(iAddress+8);
|
((u32*)VertexManager::s_pCurBufferPointer)[2] = Memory_Read_U32(iAddress+8);
|
||||||
VertexManager::s_pCurBufferPointer += 12;
|
VertexManager::s_pCurBufferPointer += 12;
|
||||||
LOG_NORMF();
|
LOG_NORMF();
|
||||||
}
|
}
|
||||||
|
@ -441,9 +441,9 @@ void LOADERDECL VertexLoader_Normal::Normal_Index16_Float3_Indices1()
|
||||||
for (int i = 0; i < 3; i++)
|
for (int i = 0; i < 3; i++)
|
||||||
{
|
{
|
||||||
u32 iAddress = arraybases[ARRAY_NORMAL] + (Index * arraystrides[ARRAY_NORMAL]) + 4*3*i;
|
u32 iAddress = arraybases[ARRAY_NORMAL] + (Index * arraystrides[ARRAY_NORMAL]) + 4*3*i;
|
||||||
((float*)VertexManager::s_pCurBufferPointer)[0] = Memory_Read_Float(iAddress);
|
((u32*)VertexManager::s_pCurBufferPointer)[0] = Memory_Read_U32(iAddress);
|
||||||
((float*)VertexManager::s_pCurBufferPointer)[1] = Memory_Read_Float(iAddress+4);
|
((u32*)VertexManager::s_pCurBufferPointer)[1] = Memory_Read_U32(iAddress+4);
|
||||||
((float*)VertexManager::s_pCurBufferPointer)[2] = Memory_Read_Float(iAddress+8);
|
((u32*)VertexManager::s_pCurBufferPointer)[2] = Memory_Read_U32(iAddress+8);
|
||||||
VertexManager::s_pCurBufferPointer += 12;
|
VertexManager::s_pCurBufferPointer += 12;
|
||||||
LOG_NORMF();
|
LOG_NORMF();
|
||||||
}
|
}
|
||||||
|
@ -484,9 +484,9 @@ void LOADERDECL VertexLoader_Normal::Normal_Index16_Float3_Indices3()
|
||||||
{
|
{
|
||||||
u16 Index = DataReadU16();
|
u16 Index = DataReadU16();
|
||||||
u32 iAddress = arraybases[ARRAY_NORMAL] + (Index * arraystrides[ARRAY_NORMAL]) + 4*3*i;
|
u32 iAddress = arraybases[ARRAY_NORMAL] + (Index * arraystrides[ARRAY_NORMAL]) + 4*3*i;
|
||||||
((float*)VertexManager::s_pCurBufferPointer)[0] = Memory_Read_Float(iAddress);
|
((u32*)VertexManager::s_pCurBufferPointer)[0] = Memory_Read_U32(iAddress);
|
||||||
((float*)VertexManager::s_pCurBufferPointer)[1] = Memory_Read_Float(iAddress+4);
|
((u32*)VertexManager::s_pCurBufferPointer)[1] = Memory_Read_U32(iAddress+4);
|
||||||
((float*)VertexManager::s_pCurBufferPointer)[2] = Memory_Read_Float(iAddress+8);
|
((u32*)VertexManager::s_pCurBufferPointer)[2] = Memory_Read_U32(iAddress+8);
|
||||||
VertexManager::s_pCurBufferPointer += 12;
|
VertexManager::s_pCurBufferPointer += 12;
|
||||||
LOG_NORMF();
|
LOG_NORMF();
|
||||||
}
|
}
|
||||||
|
|
|
@ -20,6 +20,52 @@
|
||||||
|
|
||||||
#define LOG_VTX() PRIM_LOG("vtx: %f %f %f, ", ((float*)VertexManager::s_pCurBufferPointer)[0], ((float*)VertexManager::s_pCurBufferPointer)[1], ((float*)VertexManager::s_pCurBufferPointer)[2]);
|
#define LOG_VTX() PRIM_LOG("vtx: %f %f %f, ", ((float*)VertexManager::s_pCurBufferPointer)[0], ((float*)VertexManager::s_pCurBufferPointer)[1], ((float*)VertexManager::s_pCurBufferPointer)[2]);
|
||||||
|
|
||||||
|
// Thoughts on the implementation of a vertex loader compiler.
|
||||||
|
// s_pCurBufferPointer should definitely be in a register.
|
||||||
|
// Could load the position scale factor in XMM7, for example.
|
||||||
|
|
||||||
|
// The pointer inside DataReadU8 in another.
|
||||||
|
// Let's check out Pos_ReadDirect_UByte(). For Byte, replace MOVZX with MOVSX.
|
||||||
|
|
||||||
|
/*
|
||||||
|
MOVZX(32, R(EAX), MOffset(ESI, 0));
|
||||||
|
MOVZX(32, R(EBX), MOffset(ESI, 1));
|
||||||
|
MOVZX(32, R(ECX), MOffset(ESI, 2));
|
||||||
|
MOVD(XMM0, R(EAX));
|
||||||
|
MOVD(XMM1, R(EBX));
|
||||||
|
MOVD(XMM2, R(ECX));
|
||||||
|
CVTDQ2PS(XMM0, XMM0);
|
||||||
|
CVTDQ2PS(XMM1, XMM1);
|
||||||
|
CVTDQ2PS(XMM2, XMM2);
|
||||||
|
MULSS(XMM0, XMM7);
|
||||||
|
MULSS(XMM1, XMM7);
|
||||||
|
MULSS(XMM2, XMM7);
|
||||||
|
MOVSS(MOffset(EDI, 0), XMM0);
|
||||||
|
MOVSS(MOffset(EDI, 4), XMM1);
|
||||||
|
MOVSS(MOffset(EDI, 8), XMM2);
|
||||||
|
|
||||||
|
Alternatively, lookup table:
|
||||||
|
MOVZX(32, R(EAX), MOffset(ESI, 0));
|
||||||
|
MOVZX(32, R(EBX), MOffset(ESI, 1));
|
||||||
|
MOVZX(32, R(ECX), MOffset(ESI, 2));
|
||||||
|
MOV(32, R(EAX), MComplex(LUTREG, EAX, 4));
|
||||||
|
MOV(32, R(EBX), MComplex(LUTREG, EBX, 4));
|
||||||
|
MOV(32, R(ECX), MComplex(LUTREG, ECX, 4));
|
||||||
|
MOV(MOffset(EDI, 0), XMM0);
|
||||||
|
MOV(MOffset(EDI, 4), XMM1);
|
||||||
|
MOV(MOffset(EDI, 8), XMM2);
|
||||||
|
|
||||||
|
SSE4:
|
||||||
|
PINSRB(XMM0, MOffset(ESI, 0), 0);
|
||||||
|
PINSRB(XMM0, MOffset(ESI, 1), 4);
|
||||||
|
PINSRB(XMM0, MOffset(ESI, 2), 8);
|
||||||
|
CVTDQ2PS(XMM0, XMM0);
|
||||||
|
<two unpacks here to sign extend>
|
||||||
|
MULPS(XMM0, XMM7);
|
||||||
|
MOVUPS(MOffset(EDI, 0), XMM0);
|
||||||
|
|
||||||
|
*/
|
||||||
|
|
||||||
// ==============================================================================
|
// ==============================================================================
|
||||||
// Direct
|
// Direct
|
||||||
// ==============================================================================
|
// ==============================================================================
|
||||||
|
@ -73,10 +119,11 @@ void LOADERDECL Pos_ReadDirect_Short()
|
||||||
|
|
||||||
void LOADERDECL Pos_ReadDirect_Float()
|
void LOADERDECL Pos_ReadDirect_Float()
|
||||||
{
|
{
|
||||||
((float*)VertexManager::s_pCurBufferPointer)[0] = DataReadF32();
|
// No need to use floating point here.
|
||||||
((float*)VertexManager::s_pCurBufferPointer)[1] = DataReadF32();
|
((u32 *)VertexManager::s_pCurBufferPointer)[0] = DataReadU32();
|
||||||
|
((u32 *)VertexManager::s_pCurBufferPointer)[1] = DataReadU32();
|
||||||
if (pVtxAttr->PosElements)
|
if (pVtxAttr->PosElements)
|
||||||
((float*)VertexManager::s_pCurBufferPointer)[2] = DataReadF32();
|
((u32 *)VertexManager::s_pCurBufferPointer)[2] = DataReadU32();
|
||||||
else
|
else
|
||||||
((float*)VertexManager::s_pCurBufferPointer)[2] = 1.0f;
|
((float*)VertexManager::s_pCurBufferPointer)[2] = 1.0f;
|
||||||
LOG_VTX();
|
LOG_VTX();
|
||||||
|
|
|
@ -94,15 +94,15 @@ void LOADERDECL TexCoord_ReadDirect_Short2()
|
||||||
|
|
||||||
void LOADERDECL TexCoord_ReadDirect_Float1()
|
void LOADERDECL TexCoord_ReadDirect_Float1()
|
||||||
{
|
{
|
||||||
((float*)VertexManager::s_pCurBufferPointer)[0] = DataReadF32() * tcScaleU[tcIndex];
|
((u32*)VertexManager::s_pCurBufferPointer)[0] = DataReadU32();
|
||||||
LOG_TEX1();
|
LOG_TEX1();
|
||||||
VertexManager::s_pCurBufferPointer += 4;
|
VertexManager::s_pCurBufferPointer += 4;
|
||||||
tcIndex++;
|
tcIndex++;
|
||||||
}
|
}
|
||||||
void LOADERDECL TexCoord_ReadDirect_Float2()
|
void LOADERDECL TexCoord_ReadDirect_Float2()
|
||||||
{
|
{
|
||||||
((float*)VertexManager::s_pCurBufferPointer)[0] = DataReadF32() * tcScaleU[tcIndex];
|
((u32*)VertexManager::s_pCurBufferPointer)[0] = DataReadU32();
|
||||||
((float*)VertexManager::s_pCurBufferPointer)[1] = DataReadF32() * tcScaleV[tcIndex];
|
((u32*)VertexManager::s_pCurBufferPointer)[1] = DataReadU32();
|
||||||
LOG_TEX2();
|
LOG_TEX2();
|
||||||
VertexManager::s_pCurBufferPointer += 8;
|
VertexManager::s_pCurBufferPointer += 8;
|
||||||
tcIndex++;
|
tcIndex++;
|
||||||
|
@ -201,9 +201,7 @@ void LOADERDECL TexCoord_ReadIndex8_Float1()
|
||||||
{
|
{
|
||||||
u16 Index = DataReadU8();
|
u16 Index = DataReadU8();
|
||||||
u32 iAddress = arraybases[ARRAY_TEXCOORD0+tcIndex] + (Index * arraystrides[ARRAY_TEXCOORD0+tcIndex]);
|
u32 iAddress = arraybases[ARRAY_TEXCOORD0+tcIndex] + (Index * arraystrides[ARRAY_TEXCOORD0+tcIndex]);
|
||||||
u32 uTemp;
|
((u32*)VertexManager::s_pCurBufferPointer)[0] = Memory_Read_U32(iAddress);
|
||||||
uTemp = Memory_Read_U32(iAddress);
|
|
||||||
((float*)VertexManager::s_pCurBufferPointer)[0] = *(float*)&uTemp * tcScaleU[tcIndex];
|
|
||||||
LOG_TEX1();
|
LOG_TEX1();
|
||||||
VertexManager::s_pCurBufferPointer += 4;
|
VertexManager::s_pCurBufferPointer += 4;
|
||||||
tcIndex++;
|
tcIndex++;
|
||||||
|
@ -212,11 +210,8 @@ void LOADERDECL TexCoord_ReadIndex8_Float2()
|
||||||
{
|
{
|
||||||
u16 Index = DataReadU8();
|
u16 Index = DataReadU8();
|
||||||
u32 iAddress = arraybases[ARRAY_TEXCOORD0+tcIndex] + (Index * arraystrides[ARRAY_TEXCOORD0+tcIndex]);
|
u32 iAddress = arraybases[ARRAY_TEXCOORD0+tcIndex] + (Index * arraystrides[ARRAY_TEXCOORD0+tcIndex]);
|
||||||
u32 uTemp;
|
((u32*)VertexManager::s_pCurBufferPointer)[0] = Memory_Read_U32(iAddress);
|
||||||
uTemp = Memory_Read_U32(iAddress);
|
((u32*)VertexManager::s_pCurBufferPointer)[1] = Memory_Read_U32(iAddress+4);
|
||||||
((float*)VertexManager::s_pCurBufferPointer)[0] = *(float*)&uTemp * tcScaleU[tcIndex];
|
|
||||||
uTemp = Memory_Read_U32(iAddress+4);
|
|
||||||
((float*)VertexManager::s_pCurBufferPointer)[1] = *(float*)&uTemp * tcScaleV[tcIndex];
|
|
||||||
LOG_TEX2();
|
LOG_TEX2();
|
||||||
VertexManager::s_pCurBufferPointer += 8;
|
VertexManager::s_pCurBufferPointer += 8;
|
||||||
tcIndex++;
|
tcIndex++;
|
||||||
|
@ -315,9 +310,8 @@ void LOADERDECL TexCoord_ReadIndex16_Float1()
|
||||||
{
|
{
|
||||||
u16 Index = DataReadU16();
|
u16 Index = DataReadU16();
|
||||||
u32 iAddress = arraybases[ARRAY_TEXCOORD0+tcIndex] + (Index * arraystrides[ARRAY_TEXCOORD0+tcIndex]);
|
u32 iAddress = arraybases[ARRAY_TEXCOORD0+tcIndex] + (Index * arraystrides[ARRAY_TEXCOORD0+tcIndex]);
|
||||||
u32 uTemp;
|
|
||||||
uTemp = Memory_Read_U32(iAddress );
|
((u32*)VertexManager::s_pCurBufferPointer)[0] = Memory_Read_U32(iAddress);
|
||||||
((float*)VertexManager::s_pCurBufferPointer)[0] = *(float*)&uTemp * tcScaleU[tcIndex];
|
|
||||||
LOG_TEX1();
|
LOG_TEX1();
|
||||||
VertexManager::s_pCurBufferPointer += 4;
|
VertexManager::s_pCurBufferPointer += 4;
|
||||||
tcIndex++;
|
tcIndex++;
|
||||||
|
@ -326,11 +320,9 @@ void LOADERDECL TexCoord_ReadIndex16_Float2()
|
||||||
{
|
{
|
||||||
u16 Index = DataReadU16();
|
u16 Index = DataReadU16();
|
||||||
u32 iAddress = arraybases[ARRAY_TEXCOORD0+tcIndex] + (Index * arraystrides[ARRAY_TEXCOORD0+tcIndex]);
|
u32 iAddress = arraybases[ARRAY_TEXCOORD0+tcIndex] + (Index * arraystrides[ARRAY_TEXCOORD0+tcIndex]);
|
||||||
u32 uTemp;
|
|
||||||
uTemp = Memory_Read_U32(iAddress );
|
((u32*)VertexManager::s_pCurBufferPointer)[0] = Memory_Read_U32(iAddress);
|
||||||
((float*)VertexManager::s_pCurBufferPointer)[0] = *(float*)&uTemp * tcScaleU[tcIndex];
|
((u32*)VertexManager::s_pCurBufferPointer)[1] = Memory_Read_U32(iAddress + 4);
|
||||||
uTemp = Memory_Read_U32(iAddress+4);
|
|
||||||
((float*)VertexManager::s_pCurBufferPointer)[1] = *(float*)&uTemp * tcScaleV[tcIndex];
|
|
||||||
LOG_TEX2();
|
LOG_TEX2();
|
||||||
VertexManager::s_pCurBufferPointer += 8;
|
VertexManager::s_pCurBufferPointer += 8;
|
||||||
tcIndex++;
|
tcIndex++;
|
||||||
|
|
Loading…
Reference in New Issue