| 
									
										
										
										
											2013-04-17 23:09:55 -04:00
										 |  |  | // Copyright 2013 Dolphin Emulator Project
 | 
					
						
							|  |  |  | // Licensed under GPLv2
 | 
					
						
							|  |  |  | // Refer to the license.txt file included.
 | 
					
						
							| 
									
										
										
										
											2008-12-08 05:25:12 +00:00
										 |  |  | 
 | 
					
						
							| 
									
										
										
										
											2013-02-21 02:00:27 -06:00
										 |  |  | #include <limits>
 | 
					
						
							|  |  |  | 
 | 
					
						
							| 
									
										
										
										
											2008-12-26 13:03:50 +00:00
										 |  |  | #include "Common.h"
 | 
					
						
							|  |  |  | #include "VideoCommon.h"
 | 
					
						
							| 
									
										
										
										
											2008-12-08 05:25:12 +00:00
										 |  |  | #include "VertexLoader.h"
 | 
					
						
							|  |  |  | #include "VertexLoader_Position.h"
 | 
					
						
							| 
									
										
										
										
											2010-10-03 00:41:06 +00:00
										 |  |  | #include "VertexManagerBase.h"
 | 
					
						
							| 
									
										
										
										
											2010-04-09 15:13:42 +00:00
										 |  |  | #include "CPUDetect.h"
 | 
					
						
							| 
									
										
										
										
											2008-12-08 05:25:12 +00:00
										 |  |  | 
 | 
					
						
							|  |  |  | extern float posScale; | 
					
						
							|  |  |  | extern TVtxAttr *pVtxAttr; | 
					
						
							|  |  |  | 
 | 
					
						
							|  |  |  | // Thoughts on the implementation of a vertex loader compiler.
 | 
					
						
							|  |  |  | // s_pCurBufferPointer should definitely be in a register.
 | 
					
						
							|  |  |  | // Could load the position scale factor in XMM7, for example.
 | 
					
						
							|  |  |  | 
 | 
					
						
							|  |  |  | // The pointer inside DataReadU8 in another.
 | 
					
						
							|  |  |  | // Let's check out Pos_ReadDirect_UByte(). For Byte, replace MOVZX with MOVSX.
 | 
					
						
							|  |  |  | 
 | 
					
						
							|  |  |  | /*
 | 
					
						
							|  |  |  | MOVZX(32, R(EAX), MOffset(ESI, 0)); | 
					
						
							|  |  |  | MOVZX(32, R(EBX), MOffset(ESI, 1)); | 
					
						
							|  |  |  | MOVZX(32, R(ECX), MOffset(ESI, 2)); | 
					
						
							|  |  |  | MOVD(XMM0, R(EAX)); | 
					
						
							|  |  |  | MOVD(XMM1, R(EBX)); | 
					
						
							|  |  |  | MOVD(XMM2, R(ECX));                    | 
					
						
							|  |  |  | CVTDQ2PS(XMM0, XMM0); | 
					
						
							|  |  |  | CVTDQ2PS(XMM1, XMM1); | 
					
						
							|  |  |  | CVTDQ2PS(XMM2, XMM2); | 
					
						
							|  |  |  | MULSS(XMM0, XMM7); | 
					
						
							|  |  |  | MULSS(XMM1, XMM7); | 
					
						
							|  |  |  | MULSS(XMM2, XMM7); | 
					
						
							|  |  |  | MOVSS(MOffset(EDI, 0), XMM0); | 
					
						
							|  |  |  | MOVSS(MOffset(EDI, 4), XMM1); | 
					
						
							|  |  |  | MOVSS(MOffset(EDI, 8), XMM2); | 
					
						
							|  |  |  | 
 | 
					
						
							|  |  |  | Alternatively, lookup table: | 
					
						
							|  |  |  | MOVZX(32, R(EAX), MOffset(ESI, 0)); | 
					
						
							|  |  |  | MOVZX(32, R(EBX), MOffset(ESI, 1)); | 
					
						
							|  |  |  | MOVZX(32, R(ECX), MOffset(ESI, 2)); | 
					
						
							|  |  |  | MOV(32, R(EAX), MComplex(LUTREG, EAX, 4)); | 
					
						
							|  |  |  | MOV(32, R(EBX), MComplex(LUTREG, EBX, 4)); | 
					
						
							|  |  |  | MOV(32, R(ECX), MComplex(LUTREG, ECX, 4)); | 
					
						
							|  |  |  | MOV(MOffset(EDI, 0), XMM0); | 
					
						
							|  |  |  | MOV(MOffset(EDI, 4), XMM1); | 
					
						
							|  |  |  | MOV(MOffset(EDI, 8), XMM2); | 
					
						
							|  |  |  | 
 | 
					
						
							|  |  |  | SSE4: | 
					
						
							|  |  |  | PINSRB(XMM0, MOffset(ESI, 0), 0); | 
					
						
							|  |  |  | PINSRB(XMM0, MOffset(ESI, 1), 4); | 
					
						
							|  |  |  | PINSRB(XMM0, MOffset(ESI, 2), 8); | 
					
						
							|  |  |  | CVTDQ2PS(XMM0, XMM0); | 
					
						
							|  |  |  | <two unpacks here to sign extend> | 
					
						
							|  |  |  | MULPS(XMM0, XMM7); | 
					
						
							|  |  |  | MOVUPS(MOffset(EDI, 0), XMM0); | 
					
						
							|  |  |  | 
 | 
					
						
							|  |  |  | 									 */ | 
					
						
							|  |  |  | 
 | 
					
						
							| 
									
										
										
										
											2013-02-21 02:00:27 -06:00
										 |  |  | template <typename T> | 
					
						
							|  |  |  | float PosScale(T val) | 
					
						
							| 
									
										
										
										
											2008-12-08 05:25:12 +00:00
										 |  |  | { | 
					
						
							| 
									
										
										
										
											2013-02-21 02:00:27 -06:00
										 |  |  | 	return val * posScale; | 
					
						
							| 
									
										
										
										
											2008-12-08 05:25:12 +00:00
										 |  |  | } | 
					
						
							|  |  |  | 
 | 
					
						
							| 
									
										
										
										
											2013-02-21 02:00:27 -06:00
										 |  |  | template <> | 
					
						
							|  |  |  | float PosScale(float val) | 
					
						
							| 
									
										
										
										
											2013-04-24 09:21:54 -04:00
										 |  |  | { | 
					
						
							|  |  |  | 	return val; | 
					
						
							|  |  |  | } | 
					
						
							| 
									
										
										
										
											2008-12-08 05:25:12 +00:00
										 |  |  | 
 | 
					
						
							| 
									
										
										
										
											2013-02-21 02:00:27 -06:00
										 |  |  | template <typename T, int N> | 
					
						
							| 
									
										
										
										
											2013-02-21 13:00:19 +01:00
										 |  |  | void LOADERDECL Pos_ReadDirect() | 
					
						
							| 
									
										
										
										
											2008-12-08 05:25:12 +00:00
										 |  |  | { | 
					
						
							| 
									
										
										
										
											2013-02-21 02:00:27 -06:00
										 |  |  | 	static_assert(N <= 3, "N > 3 is not sane!"); | 
					
						
							| 
									
										
										
										
											2013-03-19 21:51:12 -04:00
										 |  |  | 
 | 
					
						
							| 
									
										
										
										
											2013-02-21 13:45:48 +01:00
										 |  |  | 	for (int i = 0; i < 3; ++i) | 
					
						
							|  |  |  | 		DataWrite(i<N ? PosScale(DataRead<T>()) : 0.f); | 
					
						
							| 
									
										
										
										
											2013-03-19 21:51:12 -04:00
										 |  |  | 
 | 
					
						
							| 
									
										
										
										
											2008-12-08 05:25:12 +00:00
										 |  |  | 	LOG_VTX(); | 
					
						
							| 
									
										
										
										
											2009-02-09 21:24:32 +00:00
										 |  |  | } | 
					
						
							|  |  |  | 
 | 
					
						
							| 
									
										
										
										
											2013-02-21 02:00:27 -06:00
										 |  |  | template <typename I, typename T, int N> | 
					
						
							| 
									
										
										
										
											2013-02-21 13:00:19 +01:00
										 |  |  | void LOADERDECL Pos_ReadIndex() | 
					
						
							| 
									
										
										
										
											2009-02-15 13:45:03 +00:00
										 |  |  | { | 
					
						
							| 
									
										
										
										
											2013-02-21 02:00:27 -06:00
										 |  |  | 	static_assert(!std::numeric_limits<I>::is_signed, "Only unsigned I is sane!"); | 
					
						
							|  |  |  | 	static_assert(N <= 3, "N > 3 is not sane!"); | 
					
						
							| 
									
										
										
										
											2013-03-19 21:51:12 -04:00
										 |  |  | 
 | 
					
						
							| 
									
										
										
										
											2013-02-21 02:00:27 -06:00
										 |  |  | 	auto const index = DataRead<I>(); | 
					
						
							|  |  |  | 	if (index < std::numeric_limits<I>::max()) | 
					
						
							| 
									
										
										
										
											2010-11-14 14:42:11 +00:00
										 |  |  | 	{ | 
					
						
							| 
									
										
										
										
											2013-02-21 02:00:27 -06:00
										 |  |  | 		auto const data = reinterpret_cast<const T*>(cached_arraybases[ARRAY_POSITION] + (index * arraystrides[ARRAY_POSITION])); | 
					
						
							| 
									
										
										
										
											2013-03-19 21:51:12 -04:00
										 |  |  | 
 | 
					
						
							| 
									
										
										
										
											2013-02-21 13:45:48 +01:00
										 |  |  | 		for (int i = 0; i < 3; ++i) | 
					
						
							|  |  |  | 			DataWrite(i<N ? PosScale(Common::FromBigEndian(data[i])) : 0.f); | 
					
						
							| 
									
										
										
										
											2013-03-19 21:51:12 -04:00
										 |  |  | 
 | 
					
						
							| 
									
										
										
										
											2010-11-14 14:42:11 +00:00
										 |  |  | 		LOG_VTX(); | 
					
						
							|  |  |  | 	} | 
					
						
							| 
									
										
										
										
											2010-04-09 15:13:42 +00:00
										 |  |  | } | 
					
						
							| 
									
										
										
										
											2010-04-09 03:02:12 +00:00
										 |  |  | 
 | 
					
						
							| 
									
										
										
										
											2010-04-09 15:13:42 +00:00
										 |  |  | #if _M_SSE >= 0x301
 | 
					
						
							|  |  |  | static const __m128i kMaskSwap32_3 = _mm_set_epi32(0xFFFFFFFFL, 0x08090A0BL, 0x04050607L, 0x00010203L); | 
					
						
							|  |  |  | static const __m128i kMaskSwap32_2 = _mm_set_epi32(0xFFFFFFFFL, 0xFFFFFFFFL, 0x04050607L, 0x00010203L); | 
					
						
							| 
									
										
										
										
											2010-04-09 03:02:12 +00:00
										 |  |  | 
 | 
					
						
							| 
									
										
										
										
											2013-02-21 02:00:27 -06:00
										 |  |  | template <typename I, bool three> | 
					
						
							| 
									
										
										
										
											2013-02-21 13:00:19 +01:00
										 |  |  | void LOADERDECL Pos_ReadIndex_Float_SSSE3() | 
					
						
							| 
									
										
										
										
											2010-04-09 15:13:42 +00:00
										 |  |  | { | 
					
						
							| 
									
										
										
										
											2013-02-21 02:00:27 -06:00
										 |  |  | 	auto const index = DataRead<I>(); | 
					
						
							|  |  |  | 	if (index < std::numeric_limits<I>::max()) | 
					
						
							| 
									
										
										
										
											2010-11-14 14:42:11 +00:00
										 |  |  | 	{ | 
					
						
							| 
									
										
										
										
											2013-02-21 13:00:19 +01:00
										 |  |  | 		const u32* pData = (const u32 *)(cached_arraybases[ARRAY_POSITION] + (index * arraystrides[ARRAY_POSITION])); | 
					
						
							| 
									
										
										
										
											2011-10-08 17:33:21 +02:00
										 |  |  | 		GC_ALIGNED128(const __m128i a = _mm_loadu_si128((__m128i*)pData)); | 
					
						
							|  |  |  | 		GC_ALIGNED128(__m128i b = _mm_shuffle_epi8(a, three ? kMaskSwap32_3 : kMaskSwap32_2)); | 
					
						
							| 
									
										
										
										
											2010-11-14 14:42:11 +00:00
										 |  |  | 		_mm_storeu_si128((__m128i*)VertexManager::s_pCurBufferPointer, b); | 
					
						
							| 
									
										
										
										
											2013-02-21 02:00:27 -06:00
										 |  |  | 		VertexManager::s_pCurBufferPointer += sizeof(float) * 3; | 
					
						
							| 
									
										
										
										
											2013-02-21 13:45:48 +01:00
										 |  |  | 		LOG_VTX(); | 
					
						
							| 
									
										
										
										
											2010-11-14 14:42:11 +00:00
										 |  |  | 	} | 
					
						
							| 
									
										
										
										
											2008-12-08 05:25:12 +00:00
										 |  |  | } | 
					
						
							| 
									
										
										
										
											2010-04-09 15:13:42 +00:00
										 |  |  | #endif
 | 
					
						
							| 
									
										
										
										
											2008-12-08 05:25:12 +00:00
										 |  |  | 
 | 
					
						
							| 
									
										
										
										
											2010-04-09 15:13:42 +00:00
										 |  |  | static TPipelineFunction tableReadPosition[4][8][2] = { | 
					
						
							| 
									
										
										
										
											2010-02-28 08:41:02 +00:00
										 |  |  | 	{ | 
					
						
							|  |  |  | 		{NULL, NULL,}, | 
					
						
							|  |  |  | 		{NULL, NULL,}, | 
					
						
							|  |  |  | 		{NULL, NULL,}, | 
					
						
							|  |  |  | 		{NULL, NULL,}, | 
					
						
							|  |  |  | 		{NULL, NULL,}, | 
					
						
							|  |  |  | 	}, | 
					
						
							|  |  |  | 	{ | 
					
						
							| 
									
										
										
										
											2013-02-21 02:00:27 -06:00
										 |  |  | 		{Pos_ReadDirect<u8, 2>, Pos_ReadDirect<u8, 3>,}, | 
					
						
							|  |  |  | 		{Pos_ReadDirect<s8, 2>, Pos_ReadDirect<s8, 3>,}, | 
					
						
							|  |  |  | 		{Pos_ReadDirect<u16, 2>, Pos_ReadDirect<u16, 3>,}, | 
					
						
							|  |  |  | 		{Pos_ReadDirect<s16, 2>, Pos_ReadDirect<s16, 3>,}, | 
					
						
							|  |  |  | 		{Pos_ReadDirect<float, 2>, Pos_ReadDirect<float, 3>,}, | 
					
						
							| 
									
										
										
										
											2010-02-28 08:41:02 +00:00
										 |  |  | 	}, | 
					
						
							|  |  |  | 	{ | 
					
						
							| 
									
										
										
										
											2013-02-21 02:00:27 -06:00
										 |  |  | 		{Pos_ReadIndex<u8, u8, 2>, Pos_ReadIndex<u8, u8, 3>,}, | 
					
						
							|  |  |  | 		{Pos_ReadIndex<u8, s8, 2>, Pos_ReadIndex<u8, s8, 3>,}, | 
					
						
							|  |  |  | 		{Pos_ReadIndex<u8, u16, 2>, Pos_ReadIndex<u8, u16, 3>,}, | 
					
						
							|  |  |  | 		{Pos_ReadIndex<u8, s16, 2>, Pos_ReadIndex<u8, s16, 3>,}, | 
					
						
							|  |  |  | 		{Pos_ReadIndex<u8, float, 2>, Pos_ReadIndex<u8, float, 3>,}, | 
					
						
							| 
									
										
										
										
											2010-02-28 08:41:02 +00:00
										 |  |  | 	}, | 
					
						
							|  |  |  | 	{ | 
					
						
							| 
									
										
										
										
											2013-02-21 02:00:27 -06:00
										 |  |  | 		{Pos_ReadIndex<u16, u8, 2>, Pos_ReadIndex<u16, u8, 3>,}, | 
					
						
							|  |  |  | 		{Pos_ReadIndex<u16, s8, 2>, Pos_ReadIndex<u16, s8, 3>,}, | 
					
						
							|  |  |  | 		{Pos_ReadIndex<u16, u16, 2>, Pos_ReadIndex<u16, u16, 3>,}, | 
					
						
							|  |  |  | 		{Pos_ReadIndex<u16, s16, 2>, Pos_ReadIndex<u16, s16, 3>,}, | 
					
						
							|  |  |  | 		{Pos_ReadIndex<u16, float, 2>, Pos_ReadIndex<u16, float, 3>,}, | 
					
						
							| 
									
										
										
										
											2010-02-28 08:41:02 +00:00
										 |  |  | 	}, | 
					
						
							|  |  |  | }; | 
					
						
							|  |  |  | 
 | 
					
						
							| 
									
										
										
										
											2010-04-09 15:13:42 +00:00
										 |  |  | static int tableReadPositionVertexSize[4][8][2] = { | 
					
						
							| 
									
										
										
										
											2010-02-28 08:41:02 +00:00
										 |  |  | 	{ | 
					
						
							| 
									
										
										
										
											2013-02-21 02:00:27 -06:00
										 |  |  | 		{0, 0,}, {0, 0,}, {0, 0,}, {0, 0,}, {0, 0,}, | 
					
						
							| 
									
										
										
										
											2010-02-28 08:41:02 +00:00
										 |  |  | 	}, | 
					
						
							|  |  |  | 	{ | 
					
						
							| 
									
										
										
										
											2013-02-21 02:00:27 -06:00
										 |  |  | 		{2, 3,}, {2, 3,}, {4, 6,}, {4, 6,}, {8, 12,}, | 
					
						
							| 
									
										
										
										
											2010-02-28 08:41:02 +00:00
										 |  |  | 	}, | 
					
						
							|  |  |  | 	{ | 
					
						
							| 
									
										
										
										
											2013-02-21 02:00:27 -06:00
										 |  |  | 		{1, 1,}, {1, 1,}, {1, 1,}, {1, 1,}, {1, 1,}, | 
					
						
							| 
									
										
										
										
											2010-02-28 08:41:02 +00:00
										 |  |  | 	}, | 
					
						
							|  |  |  | 	{ | 
					
						
							| 
									
										
										
										
											2013-02-21 02:00:27 -06:00
										 |  |  | 		{2, 2,}, {2, 2,}, {2, 2,}, {2, 2,}, {2, 2,}, | 
					
						
							| 
									
										
										
										
											2010-02-28 08:41:02 +00:00
										 |  |  | 	}, | 
					
						
							|  |  |  | }; | 
					
						
							|  |  |  | 
 | 
					
						
							| 
									
										
										
										
											2010-04-09 15:13:42 +00:00
										 |  |  | 
 | 
					
						
							| 
									
										
										
										
											2013-04-24 09:21:54 -04:00
										 |  |  | void VertexLoader_Position::Init(void) | 
					
						
							|  |  |  | { | 
					
						
							| 
									
										
										
										
											2010-04-09 15:13:42 +00:00
										 |  |  | 
 | 
					
						
							|  |  |  | #if _M_SSE >= 0x301
 | 
					
						
							|  |  |  | 
 | 
					
						
							| 
									
										
										
										
											2013-04-24 09:21:54 -04:00
										 |  |  | 	if (cpu_info.bSSSE3) | 
					
						
							|  |  |  | 	{ | 
					
						
							| 
									
										
										
										
											2013-02-21 02:00:27 -06:00
										 |  |  | 		tableReadPosition[2][4][0] = Pos_ReadIndex_Float_SSSE3<u8, false>; | 
					
						
							|  |  |  | 		tableReadPosition[2][4][1] = Pos_ReadIndex_Float_SSSE3<u8, true>; | 
					
						
							|  |  |  | 		tableReadPosition[3][4][0] = Pos_ReadIndex_Float_SSSE3<u16, false>; | 
					
						
							|  |  |  | 		tableReadPosition[3][4][1] = Pos_ReadIndex_Float_SSSE3<u16, true>; | 
					
						
							| 
									
										
										
										
											2010-04-09 15:13:42 +00:00
										 |  |  | 	} | 
					
						
							|  |  |  | 
 | 
					
						
							|  |  |  | #endif
 | 
					
						
							|  |  |  | 
 | 
					
						
							|  |  |  | } | 
					
						
							|  |  |  | 
 | 
					
						
							| 
									
										
										
										
											2013-04-24 09:21:54 -04:00
										 |  |  | unsigned int VertexLoader_Position::GetSize(unsigned int _type, unsigned int _format, unsigned int _elements) | 
					
						
							|  |  |  | { | 
					
						
							| 
									
										
										
										
											2010-04-09 15:13:42 +00:00
										 |  |  | 	return tableReadPositionVertexSize[_type][_format][_elements]; | 
					
						
							|  |  |  | } | 
					
						
							|  |  |  | 
 | 
					
						
							| 
									
										
										
										
											2013-04-24 09:21:54 -04:00
										 |  |  | TPipelineFunction VertexLoader_Position::GetFunction(unsigned int _type, unsigned int _format, unsigned int _elements) | 
					
						
							|  |  |  | { | 
					
						
							| 
									
										
										
										
											2010-04-09 15:13:42 +00:00
										 |  |  | 	return tableReadPosition[_type][_format][_elements]; | 
					
						
							|  |  |  | } |