| 3226 | } |
| 3227 | |
| 3228 | Ref OpDispatchBuilder::PMADDUBSWOpImpl(IR::OpSize Size, Ref Src1, Ref Src2) { |
| 3229 | if (Size == OpSize::i64Bit) { |
| 3230 | const auto MultSize = Size << 1; |
| 3231 | // 64bit is more efficient |
| 3232 | |
| 3233 | // Src1 is unsigned |
| 3234 | auto Src1_16b = _VUXTL(MultSize, OpSize::i8Bit, Src1); // [7:0 ], [15:8], [23:16], [31:24], [39:32], [47:40], [55:48], [63:56] |
| 3235 | |
| 3236 | // Src2 is signed |
| 3237 | auto Src2_16b = _VSXTL(MultSize, OpSize::i8Bit, Src2); // [7:0 ], [15:8], [23:16], [31:24], [39:32], [47:40], [55:48], [63:56] |
| 3238 | |
| 3239 | auto ResMul_L = _VSMull(MultSize, OpSize::i16Bit, Src1_16b, Src2_16b); |
| 3240 | auto ResMul_H = _VSMull2(MultSize, OpSize::i16Bit, Src1_16b, Src2_16b); |
| 3241 | |
| 3242 | // Now add pairwise across the vector |
| 3243 | auto ResAdd = _VAddP(MultSize, OpSize::i32Bit, ResMul_L, ResMul_H); |
| 3244 | |
| 3245 | // Add saturate back down to 16bit |
| 3246 | return _VSQXTN(MultSize, OpSize::i32Bit, ResAdd); |
| 3247 | } |
| 3248 | |
| 3249 | // V{U,S}XTL{,2}/ and VUnZip{,2} can be optimized in this solution to save about one instruction. |
| 3250 | // We can up-front zero extend and sign extend the elements in-place. |
| 3251 | // This means extracting even and odd elements up-front so the unzips aren't required. |
| 3252 | // Requires implementing IR ops for BIC (vector, immediate) although. |
| 3253 | |
| 3254 | // Src1 is unsigned |
| 3255 | auto Src1_16b_L = _VUXTL(Size, OpSize::i8Bit, Src1); // [7:0 ], [15:8], [23:16], [31:24], [39:32], [47:40], [55:48], [63:56] |
| 3256 | auto Src2_16b_L = _VSXTL(Size, OpSize::i8Bit, Src2); // [7:0 ], [15:8], [23:16], [31:24], [39:32], [47:40], [55:48], [63:56] |
| 3257 | auto ResMul_L = _VMul(Size, OpSize::i16Bit, Src1_16b_L, Src2_16b_L); |
| 3258 | |
| 3259 | // Src2 is signed |
| 3260 | auto Src1_16b_H = _VUXTL2(Size, OpSize::i8Bit, Src1); // Offset to +64bits [7:0 ], [15:8], [23:16], [31:24], [39:32], [47:40], [55:48], [63:56] |
| 3261 | auto Src2_16b_H = _VSXTL2(Size, OpSize::i8Bit, Src2); // Offset to +64bits [7:0 ], [15:8], [23:16], [31:24], [39:32], [47:40], [55:48], [63:56] |
| 3262 | auto ResMul_L_H = _VMul(Size, OpSize::i16Bit, Src1_16b_H, Src2_16b_H); |
| 3263 | |
| 3264 | auto TmpZip1 = _VUnZip(Size, OpSize::i16Bit, ResMul_L, ResMul_L_H); |
| 3265 | auto TmpZip2 = _VUnZip2(Size, OpSize::i16Bit, ResMul_L, ResMul_L_H); |
| 3266 | |
| 3267 | return _VSQAdd(Size, OpSize::i16Bit, TmpZip1, TmpZip2); |
| 3268 | } |
| 3269 | |
| 3270 | void OpDispatchBuilder::PMADDUBSW(OpcodeArgs) { |
| 3271 | const auto Size = OpSizeFromSrc(Op); |
nothing calls this directly
no outgoing calls
no test coverage detected