mirror of
https://github.com/PabloMK7/citra.git
synced 2025-01-10 17:05:19 +00:00
Merge pull request #1057 from aroulin/shader-dph-dphi
Shader: Implement DPH and DPHI in interpreter/JIT
This commit is contained in:
commit
387bd3a1e4
|
@ -197,12 +197,19 @@ void RunInterpreter(UnitState<Debug>& state) {
|
||||||
|
|
||||||
case OpCode::Id::DP3:
|
case OpCode::Id::DP3:
|
||||||
case OpCode::Id::DP4:
|
case OpCode::Id::DP4:
|
||||||
|
case OpCode::Id::DPH:
|
||||||
|
case OpCode::Id::DPHI:
|
||||||
{
|
{
|
||||||
Record<DebugDataRecord::SRC1>(state.debug, iteration, src1);
|
Record<DebugDataRecord::SRC1>(state.debug, iteration, src1);
|
||||||
Record<DebugDataRecord::SRC2>(state.debug, iteration, src2);
|
Record<DebugDataRecord::SRC2>(state.debug, iteration, src2);
|
||||||
Record<DebugDataRecord::DEST_IN>(state.debug, iteration, dest);
|
Record<DebugDataRecord::DEST_IN>(state.debug, iteration, dest);
|
||||||
|
|
||||||
|
OpCode::Id opcode = instr.opcode.Value().EffectiveOpCode();
|
||||||
|
if (opcode == OpCode::Id::DPH || opcode == OpCode::Id::DPHI)
|
||||||
|
src1[3] = float24::FromFloat32(1.0f);
|
||||||
|
|
||||||
float24 dot = float24::FromFloat32(0.f);
|
float24 dot = float24::FromFloat32(0.f);
|
||||||
int num_components = (instr.opcode.Value() == OpCode::Id::DP3) ? 3 : 4;
|
int num_components = (opcode == OpCode::Id::DP3) ? 3 : 4;
|
||||||
for (int i = 0; i < num_components; ++i)
|
for (int i = 0; i < num_components; ++i)
|
||||||
dot = dot + src1[i] * src2[i];
|
dot = dot + src1[i] * src2[i];
|
||||||
|
|
||||||
|
|
|
@ -23,7 +23,7 @@ const JitFunction instr_table[64] = {
|
||||||
&JitCompiler::Compile_ADD, // add
|
&JitCompiler::Compile_ADD, // add
|
||||||
&JitCompiler::Compile_DP3, // dp3
|
&JitCompiler::Compile_DP3, // dp3
|
||||||
&JitCompiler::Compile_DP4, // dp4
|
&JitCompiler::Compile_DP4, // dp4
|
||||||
nullptr, // dph
|
&JitCompiler::Compile_DPH, // dph
|
||||||
nullptr, // unknown
|
nullptr, // unknown
|
||||||
&JitCompiler::Compile_EX2, // ex2
|
&JitCompiler::Compile_EX2, // ex2
|
||||||
&JitCompiler::Compile_LG2, // lg2
|
&JitCompiler::Compile_LG2, // lg2
|
||||||
|
@ -44,7 +44,7 @@ const JitFunction instr_table[64] = {
|
||||||
nullptr, // unknown
|
nullptr, // unknown
|
||||||
nullptr, // unknown
|
nullptr, // unknown
|
||||||
nullptr, // unknown
|
nullptr, // unknown
|
||||||
nullptr, // dphi
|
&JitCompiler::Compile_DPH, // dphi
|
||||||
nullptr, // unknown
|
nullptr, // unknown
|
||||||
&JitCompiler::Compile_SGE, // sgei
|
&JitCompiler::Compile_SGE, // sgei
|
||||||
&JitCompiler::Compile_SLT, // slti
|
&JitCompiler::Compile_SLT, // slti
|
||||||
|
@ -347,6 +347,39 @@ void JitCompiler::Compile_DP4(Instruction instr) {
|
||||||
Compile_DestEnable(instr, SRC1);
|
Compile_DestEnable(instr, SRC1);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
void JitCompiler::Compile_DPH(Instruction instr) {
|
||||||
|
if (instr.opcode.Value().EffectiveOpCode() == OpCode::Id::DPHI) {
|
||||||
|
Compile_SwizzleSrc(instr, 1, instr.common.src1i, SRC1);
|
||||||
|
Compile_SwizzleSrc(instr, 2, instr.common.src2i, SRC2);
|
||||||
|
} else {
|
||||||
|
Compile_SwizzleSrc(instr, 1, instr.common.src1, SRC1);
|
||||||
|
Compile_SwizzleSrc(instr, 2, instr.common.src2, SRC2);
|
||||||
|
}
|
||||||
|
|
||||||
|
if (Common::GetCPUCaps().sse4_1) {
|
||||||
|
// Set 4th component to 1.0
|
||||||
|
BLENDPS(SRC1, R(ONE), 0x8); // 0b1000
|
||||||
|
DPPS(SRC1, R(SRC2), 0xff);
|
||||||
|
} else {
|
||||||
|
// Reverse to set the 4th component to 1.0
|
||||||
|
SHUFPS(SRC1, R(SRC1), _MM_SHUFFLE(0, 1, 2, 3));
|
||||||
|
MOVSS(SRC1, R(ONE));
|
||||||
|
SHUFPS(SRC1, R(SRC1), _MM_SHUFFLE(0, 1, 2, 3));
|
||||||
|
|
||||||
|
MULPS(SRC1, R(SRC2));
|
||||||
|
|
||||||
|
MOVAPS(SRC2, R(SRC1));
|
||||||
|
SHUFPS(SRC1, R(SRC1), _MM_SHUFFLE(2, 3, 0, 1)); // XYZW -> ZWXY
|
||||||
|
ADDPS(SRC1, R(SRC2));
|
||||||
|
|
||||||
|
MOVAPS(SRC2, R(SRC1));
|
||||||
|
SHUFPS(SRC1, R(SRC1), _MM_SHUFFLE(0, 1, 2, 3)); // XYZW -> WZYX
|
||||||
|
ADDPS(SRC1, R(SRC2));
|
||||||
|
}
|
||||||
|
|
||||||
|
Compile_DestEnable(instr, SRC1);
|
||||||
|
}
|
||||||
|
|
||||||
void JitCompiler::Compile_EX2(Instruction instr) {
|
void JitCompiler::Compile_EX2(Instruction instr) {
|
||||||
Compile_SwizzleSrc(instr, 1, instr.common.src1, SRC1);
|
Compile_SwizzleSrc(instr, 1, instr.common.src1, SRC1);
|
||||||
MOVSS(XMM0, R(SRC1));
|
MOVSS(XMM0, R(SRC1));
|
||||||
|
|
|
@ -37,6 +37,7 @@ public:
|
||||||
void Compile_ADD(Instruction instr);
|
void Compile_ADD(Instruction instr);
|
||||||
void Compile_DP3(Instruction instr);
|
void Compile_DP3(Instruction instr);
|
||||||
void Compile_DP4(Instruction instr);
|
void Compile_DP4(Instruction instr);
|
||||||
|
void Compile_DPH(Instruction instr);
|
||||||
void Compile_EX2(Instruction instr);
|
void Compile_EX2(Instruction instr);
|
||||||
void Compile_LG2(Instruction instr);
|
void Compile_LG2(Instruction instr);
|
||||||
void Compile_MUL(Instruction instr);
|
void Compile_MUL(Instruction instr);
|
||||||
|
|
Loading…
Reference in a new issue