| 4312 | template void OpDispatchBuilder::VDPPOp<OpSize::i64Bit>(OpcodeArgs); |
| 4313 | |
| 4314 | Ref OpDispatchBuilder::MPSADBWOpImpl(IR::OpSize SrcSize, Ref Src1, Ref Src2, uint8_t Select) { |
| 4315 | const auto LaneHelper = [&, this](uint32_t Selector_Src1, uint32_t Selector_Src2, Ref Src1, Ref Src2) { |
| 4316 | // Src2 will grab a 32bit element and duplicate it across the 128bits |
| 4317 | Ref DupSrc = _VDupElement(OpSize::i128Bit, OpSize::i32Bit, Src2, Selector_Src2); |
| 4318 | |
| 4319 | // Src1/Dest needs a bunch of magic |
| 4320 | |
| 4321 | // Shift right by selected bytes |
| 4322 | // This will give us Dest[15:0], and Dest[79:64] |
| 4323 | Ref Dest1 = _VExtr(OpSize::i128Bit, OpSize::i8Bit, Src1, Src1, Selector_Src1 + 0); |
| 4324 | // This will give us Dest[31:16], and Dest[95:80] |
| 4325 | Ref Dest2 = _VExtr(OpSize::i128Bit, OpSize::i8Bit, Src1, Src1, Selector_Src1 + 1); |
| 4326 | // This will give us Dest[47:32], and Dest[111:96] |
| 4327 | Ref Dest3 = _VExtr(OpSize::i128Bit, OpSize::i8Bit, Src1, Src1, Selector_Src1 + 2); |
| 4328 | // This will give us Dest[63:48], and Dest[127:112] |
| 4329 | Ref Dest4 = _VExtr(OpSize::i128Bit, OpSize::i8Bit, Src1, Src1, Selector_Src1 + 3); |
| 4330 | |
| 4331 | // For each shifted section, we now have two 32-bit values per vector that can be used |
| 4332 | // Dest1.S[0] and Dest1.S[1] = Bytes - 0,1,2,3:4,5,6,7 |
| 4333 | // Dest2.S[0] and Dest2.S[1] = Bytes - 1,2,3,4:5,6,7,8 |
| 4334 | // Dest3.S[0] and Dest3.S[1] = Bytes - 2,3,4,5:6,7,8,9 |
| 4335 | // Dest4.S[0] and Dest4.S[1] = Bytes - 3,4,5,6:7,8,9,10 |
| 4336 | Dest1 = _VUABDL(OpSize::i128Bit, OpSize::i8Bit, Dest1, DupSrc); |
| 4337 | Dest2 = _VUABDL(OpSize::i128Bit, OpSize::i8Bit, Dest2, DupSrc); |
| 4338 | Dest3 = _VUABDL(OpSize::i128Bit, OpSize::i8Bit, Dest3, DupSrc); |
| 4339 | Dest4 = _VUABDL(OpSize::i128Bit, OpSize::i8Bit, Dest4, DupSrc); |
| 4340 | |
| 4341 | // Dest[1,2,3,4] Now contains the data prior to combining |
| 4342 | // Temp[0,1,2,3] for each step |
| 4343 | |
| 4344 | // Each destination now has 16bit x 8 elements in it that were the absolute difference for each byte |
| 4345 | // Needs each to be 16bit to store the next step |
| 4346 | // Next stage is to sum pairwise |
| 4347 | // Dest1: |
| 4348 | // ADDP Dest3, Dest1: TmpCombine1 |
| 4349 | // ADDP Dest4, Dest2: TmpCombine2 |
| 4350 | // TmpCombine1.8H[0] = Dest1.8H[0] + Dest1.8H[1]; |
| 4351 | // TmpCombine1.8H[1] = Dest1.8H[2] + Dest1.8H[3]; |
| 4352 | // TmpCombine1.8H[2] = Dest1.8H[4] + Dest1.8H[5]; |
| 4353 | // TmpCombine1.8H[3] = Dest1.8H[6] + Dest1.8H[7]; |
| 4354 | // TmpCombine1.8H[4] = Dest3.8H[0] + Dest3.8H[1]; |
| 4355 | // TmpCombine1.8H[5] = Dest3.8H[2] + Dest3.8H[3]; |
| 4356 | // TmpCombine1.8H[6] = Dest3.8H[4] + Dest3.8H[5]; |
| 4357 | // TmpCombine1.8H[7] = Dest3.8H[6] + Dest3.8H[7]; |
| 4358 | // <Repeat for Dest4 and Dest3> |
| 4359 | auto TmpCombine1 = _VAddP(OpSize::i128Bit, OpSize::i16Bit, Dest1, Dest3); |
| 4360 | auto TmpCombine2 = _VAddP(OpSize::i128Bit, OpSize::i16Bit, Dest2, Dest4); |
| 4361 | |
| 4362 | // TmpTranspose1: |
| 4363 | // VTrn TmpCombine1, TmpCombine2: TmpTranspose1 |
| 4364 | // Transposes Even and odd elements so we can use vaddp for final results. |
| 4365 | auto TmpTranspose1 = _VTrn(OpSize::i128Bit, OpSize::i32Bit, TmpCombine1, TmpCombine2); |
| 4366 | auto TmpTranspose2 = _VTrn2(OpSize::i128Bit, OpSize::i32Bit, TmpCombine1, TmpCombine2); |
| 4367 | |
| 4368 | // ADDP TmpTranspose1, TmpTranspose2: FinalCombine |
| 4369 | // FinalCombine.8H[0] = TmpTranspose1.8H[0] + TmpTranspose1.8H[1] |
| 4370 | // FinalCombine.8H[1] = TmpTranspose1.8H[2] + TmpTranspose1.8H[3] |
| 4371 | // FinalCombine.8H[2] = TmpTranspose1.8H[4] + TmpTranspose1.8H[5] |
nothing calls this directly
no outgoing calls
no test coverage detected