| 3327 | template void OpDispatchBuilder::VPMULHWOp<true>(OpcodeArgs); |
| 3328 | |
| 3329 | Ref OpDispatchBuilder::PMULHRSWOpImpl(OpSize Size, Ref Src1, Ref Src2) { |
| 3330 | Ref Res {}; |
| 3331 | if (Size == OpSize::i64Bit) { |
| 3332 | // Implementation is more efficient for 8byte registers |
| 3333 | Res = _VSMull(Size << 1, OpSize::i16Bit, Src1, Src2); |
| 3334 | Res = _VSShrI(Size << 1, OpSize::i32Bit, Res, 14); |
| 3335 | auto OneVector = _VectorImm(Size << 1, OpSize::i32Bit, 1); |
| 3336 | Res = _VAdd(Size << 1, OpSize::i32Bit, Res, OneVector); |
| 3337 | return _VUShrNI(Size << 1, OpSize::i32Bit, Res, 1); |
| 3338 | } else { |
| 3339 | // 128-bit and 256-bit are less efficient |
| 3340 | Ref ResultLow; |
| 3341 | Ref ResultHigh; |
| 3342 | |
| 3343 | ResultLow = _VSMull(Size, OpSize::i16Bit, Src1, Src2); |
| 3344 | ResultHigh = _VSMull2(Size, OpSize::i16Bit, Src1, Src2); |
| 3345 | |
| 3346 | ResultLow = _VSShrI(Size, OpSize::i32Bit, ResultLow, 14); |
| 3347 | ResultHigh = _VSShrI(Size, OpSize::i32Bit, ResultHigh, 14); |
| 3348 | auto OneVector = _VectorImm(Size, OpSize::i32Bit, 1); |
| 3349 | |
| 3350 | ResultLow = _VAdd(Size, OpSize::i32Bit, ResultLow, OneVector); |
| 3351 | ResultHigh = _VAdd(Size, OpSize::i32Bit, ResultHigh, OneVector); |
| 3352 | |
| 3353 | // Combine the results |
| 3354 | Res = _VUShrNI(Size, OpSize::i32Bit, ResultLow, 1); |
| 3355 | return _VUShrNI2(Size, OpSize::i32Bit, Res, ResultHigh, 1); |
| 3356 | } |
| 3357 | } |
| 3358 | |
| 3359 | void OpDispatchBuilder::PMULHRSW(OpcodeArgs) { |
| 3360 | Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags); |
nothing calls this directly
no outgoing calls
no test coverage detected