MCPcopy Create free account
hub / github.com/FEX-Emu/FEX / MPSADBWOpImpl

Method MPSADBWOpImpl

FEXCore/Source/Interface/Core/OpcodeDispatcher/Vector.cpp:4314–4399  ·  view source on GitHub ↗

Source from the content-addressed store, hash-verified

4312template void OpDispatchBuilder::VDPPOp<OpSize::i64Bit>(OpcodeArgs);
4313
4314Ref OpDispatchBuilder::MPSADBWOpImpl(IR::OpSize SrcSize, Ref Src1, Ref Src2, uint8_t Select) {
4315 const auto LaneHelper = [&, this](uint32_t Selector_Src1, uint32_t Selector_Src2, Ref Src1, Ref Src2) {
4316 // Src2 will grab a 32bit element and duplicate it across the 128bits
4317 Ref DupSrc = _VDupElement(OpSize::i128Bit, OpSize::i32Bit, Src2, Selector_Src2);
4318
4319 // Src1/Dest needs a bunch of magic
4320
4321 // Shift right by selected bytes
4322 // This will give us Dest[15:0], and Dest[79:64]
4323 Ref Dest1 = _VExtr(OpSize::i128Bit, OpSize::i8Bit, Src1, Src1, Selector_Src1 + 0);
4324 // This will give us Dest[31:16], and Dest[95:80]
4325 Ref Dest2 = _VExtr(OpSize::i128Bit, OpSize::i8Bit, Src1, Src1, Selector_Src1 + 1);
4326 // This will give us Dest[47:32], and Dest[111:96]
4327 Ref Dest3 = _VExtr(OpSize::i128Bit, OpSize::i8Bit, Src1, Src1, Selector_Src1 + 2);
4328 // This will give us Dest[63:48], and Dest[127:112]
4329 Ref Dest4 = _VExtr(OpSize::i128Bit, OpSize::i8Bit, Src1, Src1, Selector_Src1 + 3);
4330
4331 // For each shifted section, we now have two 32-bit values per vector that can be used
4332 // Dest1.S[0] and Dest1.S[1] = Bytes - 0,1,2,3:4,5,6,7
4333 // Dest2.S[0] and Dest2.S[1] = Bytes - 1,2,3,4:5,6,7,8
4334 // Dest3.S[0] and Dest3.S[1] = Bytes - 2,3,4,5:6,7,8,9
4335 // Dest4.S[0] and Dest4.S[1] = Bytes - 3,4,5,6:7,8,9,10
4336 Dest1 = _VUABDL(OpSize::i128Bit, OpSize::i8Bit, Dest1, DupSrc);
4337 Dest2 = _VUABDL(OpSize::i128Bit, OpSize::i8Bit, Dest2, DupSrc);
4338 Dest3 = _VUABDL(OpSize::i128Bit, OpSize::i8Bit, Dest3, DupSrc);
4339 Dest4 = _VUABDL(OpSize::i128Bit, OpSize::i8Bit, Dest4, DupSrc);
4340
4341 // Dest[1,2,3,4] Now contains the data prior to combining
4342 // Temp[0,1,2,3] for each step
4343
4344 // Each destination now has 16bit x 8 elements in it that were the absolute difference for each byte
4345 // Needs each to be 16bit to store the next step
4346 // Next stage is to sum pairwise
4347 // Dest1:
4348 // ADDP Dest3, Dest1: TmpCombine1
4349 // ADDP Dest4, Dest2: TmpCombine2
4350 // TmpCombine1.8H[0] = Dest1.8H[0] + Dest1.8H[1];
4351 // TmpCombine1.8H[1] = Dest1.8H[2] + Dest1.8H[3];
4352 // TmpCombine1.8H[2] = Dest1.8H[4] + Dest1.8H[5];
4353 // TmpCombine1.8H[3] = Dest1.8H[6] + Dest1.8H[7];
4354 // TmpCombine1.8H[4] = Dest3.8H[0] + Dest3.8H[1];
4355 // TmpCombine1.8H[5] = Dest3.8H[2] + Dest3.8H[3];
4356 // TmpCombine1.8H[6] = Dest3.8H[4] + Dest3.8H[5];
4357 // TmpCombine1.8H[7] = Dest3.8H[6] + Dest3.8H[7];
4358 // <Repeat for Dest4 and Dest3>
4359 auto TmpCombine1 = _VAddP(OpSize::i128Bit, OpSize::i16Bit, Dest1, Dest3);
4360 auto TmpCombine2 = _VAddP(OpSize::i128Bit, OpSize::i16Bit, Dest2, Dest4);
4361
4362 // TmpTranspose1:
4363 // VTrn TmpCombine1, TmpCombine2: TmpTranspose1
4364 // Transposes Even and odd elements so we can use vaddp for final results.
4365 auto TmpTranspose1 = _VTrn(OpSize::i128Bit, OpSize::i32Bit, TmpCombine1, TmpCombine2);
4366 auto TmpTranspose2 = _VTrn2(OpSize::i128Bit, OpSize::i32Bit, TmpCombine1, TmpCombine2);
4367
4368 // ADDP TmpTranspose1, TmpTranspose2: FinalCombine
4369 // FinalCombine.8H[0] = TmpTranspose1.8H[0] + TmpTranspose1.8H[1]
4370 // FinalCombine.8H[1] = TmpTranspose1.8H[2] + TmpTranspose1.8H[3]
4371 // FinalCombine.8H[2] = TmpTranspose1.8H[4] + TmpTranspose1.8H[5]

Callers

nothing calls this directly

Calls

no outgoing calls

Tested by

no test coverage detected